test(integration): exact four-part translation cases on a shared fake provider and shared YAML deployment (#44451)

* test(integration): exact four-part translation cases on a shared fake provider and shared YAML deployment

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* test(integration): compare every non-transport provider header and check for late provider requests at session end

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* test(integration): name TranslationTestCase fields after litellm and provider sides and drop regressions

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* test(integration): prefix checked TranslationTestCase fields with expected_ and name the fake reply mock_provider_response

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* docs(integration): name TranslationTestCase fields in the translation README

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* test(integration): add a claude-opus-5-5 base case and deployment next to claude-sonnet-4-6

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* test(integration): name translation cases <MODEL>_TEST_CASE and document the naming rule

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* docs(integration): move translation test rules into tests/integration/translation

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

---------

Co-authored-by: kerry <kerry@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-04 00:09:58 +00:00 • committed by GitHub
parent 0ed1c08f02
commit f850b2c324
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
19 changed files with 385 additions and 2 deletions

View file

@ -20,6 +20,8 @@ There is no per-node manifest. A positional argument is a file of the group or a
Provider sentinels currently use the controlled server, not live recordings. The provider shard also runs the existing strict replay controls for changed requests, exhausted interactions, leftover interactions and no provider connection. Future recorded scenarios must use that replay-only implementation; missing recordings cannot fall back to a real provider. The observation endpoint is destructive and the current selection runs serially against one owned upstream
Translation tests in `translation/` compare exact provider requests and LiteLLM responses against a shared fake provider; `translation/README.md` has their rules
Fixtures must contain synthetic data only. Keep private incident records and source documents out of code, fixtures, logs and PR descriptions
Database cases own their temporary schemas, roles, constraints and proxy processes. They prove reader-versus-writer execution with PostgreSQL lock observations, exercise real transaction wait limits and verify rollback after a reached database failure

View file

@ -11,6 +11,7 @@ OWNED_DIRECTORIES: Final = frozenset(
"providers",
"streaming",
"messages_endpoint",
"translation",
"configuration",
"mcp",
"observability",

View file

@ -0,0 +1,44 @@
"""The fake provider shared by every integration test that takes the `provider` fixture.
Deployments in `proxy_config.yaml` point at `PROVIDER_URL`, so one server answers for all of them. A test
queues the replies it expects with `expect` and reads what the proxy sent with `received`. Tests run one at a
time against it; the `provider` fixture checks nothing is left over between tests.
"""
from __future__ import annotations
from collections import deque
from collections.abc import Iterator
from contextlib import contextmanager
from dataclasses import dataclass, field
from typing import Final
from tests.integration._support.wire import Reply, Request, Wire, wire_server
PROVIDER_PORT: Final = 8191
PROVIDER_URL: Final = f"http://127.0.0.1:{PROVIDER_PORT}"
_UNQUEUED: Final = Reply(status=500, body=b'{"error": "the shared fake provider has no reply queued for this request"}')
@dataclass(slots=True)
class SharedProvider:
wire: Wire
replies: deque[Reply]
last_test: str | None = field(default=None)
def expect(self, *replies: Reply) -> None:
self.replies.extend(replies)
def received(self) -> tuple[Request, ...]:
return self.wire.drain()
@contextmanager
def shared_provider() -> Iterator[SharedProvider]:
replies: Final[deque[Reply]] = deque()
def respond(request: Request) -> Reply:
return replies.popleft() if replies else _UNQUEUED
with wire_server(respond, port=PROVIDER_PORT) as wire:
yield SharedProvider(wire, replies)

View file

@ -15,6 +15,7 @@ from redis import Redis
from tests.integration._support.client import Gateway, eventually, gateway_from_environment
from tests.integration._support.generation import LIFECYCLE_SETTINGS
from tests.integration._support.manifest import OWNED_DIRECTORIES
from tests.integration._support.provider import SharedProvider, shared_provider
from tests.integration._support.routing import RoutingPlugin
from tests.integration.run import GITHUB_FILES
@ -130,6 +131,33 @@ def gateway() -> Iterator[Gateway]:
yield value
@pytest.fixture(scope="session")
def shared_provider_server() -> Iterator[SharedProvider]:
if os.environ.get("PYTEST_XDIST_WORKER"):
pytest.fail("the shared fake provider needs tests to run one at a time; this group runs under pytest-xdist")
with shared_provider() as server:
yield server
late: Final = server.received()
assert late == (), f"the shared fake provider got {[item.target for item in late]} after {server.last_test} finished"
@pytest.fixture
def provider(shared_provider_server: SharedProvider, request: pytest.FixtureRequest) -> Iterator[SharedProvider]:
stray: Final = shared_provider_server.received()
shared_provider_server.replies.clear()
assert stray == (), (
f"the shared fake provider got {[item.target for item in stray]} "
f"after {shared_provider_server.last_test} finished"
)
yield shared_provider_server
shared_provider_server.last_test = request.node.nodeid
unused: Final = len(shared_provider_server.replies)
unread: Final = shared_provider_server.received()
shared_provider_server.replies.clear()
assert unused == 0, f"{unused} queued provider replies were never requested"
assert unread == (), f"the test never read the provider requests {[item.target for item in unread]}"
@pytest.fixture
def peer(gateway: Gateway) -> Iterator[Gateway]:
url: Final = os.environ["INTEGRATION_PEER_URL"]

View file

@ -1,4 +1,14 @@
model_list: []
model_list:
- model_name: anthropic/claude-opus-5-5
litellm_params:
model: anthropic/claude-opus-5-5
api_base: http://127.0.0.1:8191
api_key: synthetic-anthropic-key
- model_name: anthropic/claude-sonnet-4-6
litellm_params:
model: anthropic/claude-sonnet-4-6
api_base: http://127.0.0.1:8191
api_key: synthetic-anthropic-key
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
database_url: os.environ/DATABASE_URL

View file

@ -15,7 +15,7 @@ GROUPS: Final = MappingProxyType(
"management": ("management", "authorization", "configuration"),
"accounting": ("pricing", "spend"),
"database": ("database",),
"providers": ("providers", "routing", "streaming", "messages_endpoint"),
"providers": ("providers", "routing", "streaming", "messages_endpoint", "translation"),
"extensions": ("observability", "compatibility"),
"mcp": ("mcp",),
"sdk": ("sdk",),

View file

@ -0,0 +1,21 @@
# tests/integration/translation
Exact translation cases on the shared fake provider. `README.md` here has the case fields, deployments and capture steps
## Naming
Each model gets one complete base `TranslationTestCase` in `translation/<endpoint>/bases/<provider>.py`,
named `<MODEL>_TEST_CASE` after its deployment (`anthropic/claude-sonnet-4-6` is
`CLAUDE_SONNET_4_6_TEST_CASE`). A feature case is `<MODEL>_<SCENARIO>_TEST_CASE`. Import a base under its
own name, never aliased to `BASE`, so every case shows which model it derives from
```python
from integration.translation.messages.bases.anthropic import CLAUDE_SONNET_4_6_TEST_CASE
CLAUDE_SONNET_4_6_THINKING_BUDGET_TEST_CASE: Final = replace(
CLAUDE_SONNET_4_6_TEST_CASE,
scenario="thinking_budget",
litellm_request={**CLAUDE_SONNET_4_6_TEST_CASE.litellm_request, "max_tokens": 2048, "thinking": ...},
...
)
```

View file

@ -0,0 +1,7 @@
# Translation tests
These tests check one request through the proxy as literals on a `TranslationTestCase`: the `litellm_endpoint` and `litellm_request` the test sends, the `expected_provider_endpoint`, `expected_provider_headers` and `expected_provider_request` the fake provider must receive, the `mock_provider_response` it answers with, and the `expected_litellm_status_code` and `expected_litellm_response` the test must get back. The runner compares the provider request body, every provider header other than transport headers, and the LiteLLM response body in full, so an added, removed, renamed or moved field fails. Folders follow the client endpoint and then the feature, for example `translation/messages/reasoning/`. Each endpoint keeps one complete base case per model in `<endpoint>/bases/<provider>.py`, named `<MODEL>_TEST_CASE` (for example `CLAUDE_SONNET_4_6_TEST_CASE`), which `<endpoint>/basic/` runs on its own. A feature case is `<MODEL>_<SCENARIO>_TEST_CASE = dataclasses.replace(<MODEL>_TEST_CASE, ...)`, imports the base under its own name rather than as `BASE`, and lists only the fields it changes
Deployments used by translation tests are shared by the whole suite and declared in `proxy_config.yaml` under `model_list`, with `model_name` equal to the litellm model string, `api_base: http://127.0.0.1:8191` and a synthetic key. A case names the deployment literally in its client request. The fake provider behind them is the `provider` fixture: one `wire_server` on port 8191 inside the pytest process, started the first time a test asks for it. A test queues its replies with `provider.expect(...)` and reads what the proxy sent with `provider.received()`. Tests that use it run one at a time. After each test the fixture fails if a queued reply was never requested or a received request was never read, and before each test it fails if a request arrived in between, naming the previous test. It refuses to start under pytest-xdist, so the `mcp` and `cost` groups cannot use it. Client requests carry `"cache": {"no-cache": True}` because the proxy caches responses in Redis
Provider responses in translation cases are captured once from the real provider and stored verbatim. First run the new case against the fake provider; the body the proxy sends is the case's `expected_provider_request`. Send that exact body to the real provider endpoint with a key from the 1Password `Shared` vault (`/qa-keys`), and only store the case when the provider answers 2xx. Keep only the response body and drop every response header, since headers carry account identifiers such as the organization id and rate limits. Never print or save the request headers you sent. Before committing, check the body contains no key and no account identifier, such as an organization id, an AWS account id inside an ARN, a GCP project id or an Azure resource name, and that the prompt is synthetic. Paste the body as the case's `mock_provider_response` without shortening ids, token counts or signatures, and move long opaque values such as thinking signatures into module-level constants used by both `mock_provider_response` and `expected_litellm_response`. Do not commit the script used for the capture

View file

@ -0,0 +1,24 @@
from collections.abc import Mapping
from dataclasses import dataclass
from pydantic import JsonValue
@dataclass(frozen=True, slots=True, kw_only=True)
class TranslationTestCase:
"""One request through the proxy to a deployment in `proxy_config.yaml`: what the test sends to LiteLLM, the
exact request the provider must receive, the fake provider's reply, and the exact response LiteLLM must return."""
scenario: str
litellm_endpoint: str
litellm_request: Mapping[str, JsonValue]
expected_provider_endpoint: str
expected_provider_headers: Mapping[str, str]
expected_provider_request: Mapping[str, JsonValue]
mock_provider_response: Mapping[str, JsonValue]
expected_litellm_status_code: int = 200
expected_litellm_response: Mapping[str, JsonValue]
@property
def id(self) -> str:
return f"{self.litellm_request['model']}-{self.scenario}"

View file

@ -0,0 +1,3 @@
import pytest
pytest.register_assert_rewrite("integration.translation.runner")

View file

@ -0,0 +1,139 @@
from typing import Final
from integration.translation.case import TranslationTestCase
CLAUDE_OPUS_5_5_TEST_CASE: Final = TranslationTestCase(
scenario="basic",
litellm_endpoint="/v1/messages",
litellm_request={
"model": "anthropic/claude-opus-5-5",
"max_tokens": 64,
"system": "You are a terse assistant.",
"messages": [{"role": "user", "content": "Say hello."}],
"cache": {"no-cache": True},
},
expected_provider_endpoint="/v1/messages",
expected_provider_headers={
"x-api-key": "synthetic-anthropic-key",
"anthropic-version": "2023-06-01",
"content-type": "application/json",
},
expected_provider_request={
"model": "claude-opus-5-5",
"max_tokens": 64,
"stream": False,
"system": "You are a terse assistant.",
"messages": [{"role": "user", "content": "Say hello."}],
},
mock_provider_response={
"model": "claude-opus-5-5",
"id": "msg_011CfgC5HKvyve78CTFAw97f",
"type": "message",
"role": "assistant",
"content": [{"type": "text", "text": "Hello."}],
"container": None,
"stop_reason": "end_turn",
"stop_sequence": None,
"stop_details": None,
"usage": {
"input_tokens": 23,
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"cache_creation": {"ephemeral_5m_input_tokens": 0, "ephemeral_1h_input_tokens": 0},
"output_tokens": 6,
"output_tokens_details": {"thinking_tokens": 0},
"service_tier": "standard",
"inference_geo": "global",
},
"diagnostics": None,
},
expected_litellm_response={
"model": "anthropic/claude-opus-5-5",
"id": "msg_011CfgC5HKvyve78CTFAw97f",
"type": "message",
"role": "assistant",
"content": [{"type": "text", "text": "Hello."}],
"container": None,
"stop_reason": "end_turn",
"stop_sequence": None,
"stop_details": None,
"usage": {
"input_tokens": 23,
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"cache_creation": {"ephemeral_5m_input_tokens": 0, "ephemeral_1h_input_tokens": 0},
"output_tokens": 6,
"output_tokens_details": {"thinking_tokens": 0},
"service_tier": "standard",
"inference_geo": "global",
},
"diagnostics": None,
},
)
CLAUDE_SONNET_4_6_TEST_CASE: Final = TranslationTestCase(
scenario="basic",
litellm_endpoint="/v1/messages",
litellm_request={
"model": "anthropic/claude-sonnet-4-6",
"max_tokens": 64,
"system": "You are a terse assistant.",
"messages": [{"role": "user", "content": "Say hello."}],
"cache": {"no-cache": True},
},
expected_provider_endpoint="/v1/messages",
expected_provider_headers={
"x-api-key": "synthetic-anthropic-key",
"anthropic-version": "2023-06-01",
"content-type": "application/json",
},
expected_provider_request={
"model": "claude-sonnet-4-6",
"max_tokens": 64,
"stream": False,
"system": "You are a terse assistant.",
"messages": [{"role": "user", "content": "Say hello."}],
},
mock_provider_response={
"model": "claude-sonnet-4-6",
"id": "msg_011CffzUNHaEfzVxCh5hskBG",
"type": "message",
"role": "assistant",
"content": [{"type": "text", "text": "Hello!"}],
"container": None,
"stop_reason": "end_turn",
"stop_sequence": None,
"stop_details": None,
"usage": {
"input_tokens": 18,
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"cache_creation": {"ephemeral_5m_input_tokens": 0, "ephemeral_1h_input_tokens": 0},
"output_tokens": 5,
"service_tier": "standard",
"inference_geo": "global",
},
"diagnostics": None,
},
expected_litellm_response={
"model": "anthropic/claude-sonnet-4-6",
"id": "msg_011CffzUNHaEfzVxCh5hskBG",
"type": "message",
"role": "assistant",
"content": [{"type": "text", "text": "Hello!"}],
"container": None,
"stop_reason": "end_turn",
"stop_sequence": None,
"stop_details": None,
"usage": {
"input_tokens": 18,
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"cache_creation": {"ephemeral_5m_input_tokens": 0, "ephemeral_1h_input_tokens": 0},
"output_tokens": 5,
"service_tier": "standard",
"inference_geo": "global",
},
"diagnostics": None,
},
)

View file

@ -0,0 +1,11 @@
import pytest
from integration._support.client import Gateway
from integration._support.provider import SharedProvider
from integration.translation.case import TranslationTestCase
from integration.translation.messages.bases.anthropic import CLAUDE_OPUS_5_5_TEST_CASE, CLAUDE_SONNET_4_6_TEST_CASE
from integration.translation.runner import run
@pytest.mark.parametrize("case", [CLAUDE_OPUS_5_5_TEST_CASE, CLAUDE_SONNET_4_6_TEST_CASE], ids=lambda case: case.id)
def test_messages_basic_anthropic(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
run(case, gateway, provider)

View file

@ -0,0 +1,72 @@
from dataclasses import replace
from typing import Final
import pytest
from integration._support.client import Gateway
from integration._support.provider import SharedProvider
from integration.translation.case import TranslationTestCase
from integration.translation.messages.bases.anthropic import CLAUDE_SONNET_4_6_TEST_CASE
from integration.translation.runner import run
SIGNATURE_1: Final = (
"EpECCqgBCBIYAipAivUPApu85FYYe3+cXal8EiJOza7QGqKyekC8vDSn4oyeqGa2CrarO4abiuG7dzBXjmYR8+daw4h50ZjKmak7czIRY2xh"
"dWRlLXNvbm5ldC00LTY4AEIIdGhpbmtpbmdaJGQwMDgxZjJiLWQ5NjEtNGFhYi05ZTRjLTcxYmU3ZTA0ZTY3MJoBEwoRY2xhdWRlLXNvbm5l"
"dC00LTaoAY3fhdYGEgwJHsjNkCTlV9k1jWQaDOmeP/z67YtLTojSqCIwhRXGrNzSuGfMD1HqA72lctQCy83Wkr0u8W5lBXXn+MD6WfJGTJqM"
"1FW7qRmOMOKJKhbheeMpsTs7XvdvsiDQqgM4PAJt4cwgGAE="
)
CLAUDE_SONNET_4_6_THINKING_BUDGET_TEST_CASE: Final = replace(
CLAUDE_SONNET_4_6_TEST_CASE,
scenario="thinking_budget",
litellm_request={
**CLAUDE_SONNET_4_6_TEST_CASE.litellm_request,
"max_tokens": 2048,
"thinking": {"type": "enabled", "budget_tokens": 1024},
},
expected_provider_request={
**CLAUDE_SONNET_4_6_TEST_CASE.expected_provider_request,
"max_tokens": 2048,
"thinking": {"type": "enabled", "budget_tokens": 1024},
},
mock_provider_response={
**CLAUDE_SONNET_4_6_TEST_CASE.mock_provider_response,
"id": "msg_011CffzUREgTzMm1dXRqP2LR",
"content": [
{"type": "thinking", "thinking": "Hello!", "signature": SIGNATURE_1},
{"type": "text", "text": "Hello!"},
],
"usage": {
"input_tokens": 47,
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"cache_creation": {"ephemeral_5m_input_tokens": 0, "ephemeral_1h_input_tokens": 0},
"output_tokens": 15,
"output_tokens_details": {"thinking_tokens": 7},
"service_tier": "standard",
"inference_geo": "global",
},
},
expected_litellm_response={
**CLAUDE_SONNET_4_6_TEST_CASE.expected_litellm_response,
"id": "msg_011CffzUREgTzMm1dXRqP2LR",
"content": [
{"type": "thinking", "thinking": "Hello!", "signature": SIGNATURE_1},
{"type": "text", "text": "Hello!"},
],
"usage": {
"input_tokens": 47,
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"cache_creation": {"ephemeral_5m_input_tokens": 0, "ephemeral_1h_input_tokens": 0},
"output_tokens": 15,
"output_tokens_details": {"thinking_tokens": 7},
"service_tier": "standard",
"inference_geo": "global",
},
},
)
@pytest.mark.parametrize("case", [CLAUDE_SONNET_4_6_THINKING_BUDGET_TEST_CASE], ids=lambda case: case.id)
def test_messages_reasoning_anthropic(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
run(case, gateway, provider)

View file

@ -0,0 +1,21 @@
import json
from typing import Final
from integration._support.client import Gateway
from integration._support.provider import SharedProvider
from integration._support.wire import Reply
from integration.translation.case import TranslationTestCase
TRANSPORT_HEADERS: Final = frozenset({"host", "accept", "accept-encoding", "connection", "content-length", "user-agent"})
def run(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
provider.expect(Reply(body=json.dumps(case.mock_provider_response).encode()))
response: Final = gateway.request("POST", case.litellm_endpoint, case.litellm_request)
received: Final = provider.received()
assert [(request.method, request.target) for request in received] == [("POST", case.expected_provider_endpoint)]
sent: Final = received[0]
assert {name: value for name, value in sent.headers.items() if name not in TRANSPORT_HEADERS} == dict(case.expected_provider_headers)
assert json.loads(sent.body) == case.expected_provider_request
assert response.status_code == case.expected_litellm_status_code, response.text
assert response.json() == case.expected_litellm_response