test(e2e): retry timeout-shaped Mantle test_connection probes

The endpoint answers a probe that exceeds HEALTH_CHECK_TIMEOUT_SECONDS with
HTTP 200 and an in-body "Timeout exceeded", which the harness's status-code
rerun policy cannot see. The suite's parallel Bedrock load can push a Mantle
probe past that cap transiently, so only that exact error is retried, three
bounded attempts with visible prints; any other error verdict still fails
immediately.
This commit is contained in:
mateo-berri 2026-08-25 11:05:38 -07:00
parent 4b5e3db890
commit 90f9a8bfda

View file

@ -8,35 +8,60 @@ verdict from the live provider rather than just a 200 envelope. The region is a
literal because the endpoint rejects request-supplied os.environ/ references;
credentials fall through to the proxy's own environment (bearer token locally,
pod identity in CI).
The endpoint caps every probe at HEALTH_CHECK_TIMEOUT_SECONDS and answers a
timed-out probe with HTTP 200 and an in-body "Timeout exceeded", which the
harness's status-code retry policy cannot see. A Mantle probe can hit that cap
transiently while the rest of the suite saturates the same AWS account, so only
that exact error is retried here; any other error verdict fails immediately.
"""
from __future__ import annotations
import time
import pytest
from e2e_http import unwrap
from management_client import ManagementClient
from models import ConnectionTestBody, LiteLLMParamsBody
from models import ConnectionTestBody, ConnectionTestResponse, LiteLLMParamsBody
pytestmark = pytest.mark.e2e
MANTLE_RESPONSES_BACKEND = "bedrock_mantle/openai.gpt-5.6-luna"
MANTLE_REGION = "us-east-1"
PROBE_TIMEOUT_ERROR = "Timeout exceeded"
PROBE_ATTEMPTS = 3
PROBE_RETRY_SLEEP_SECONDS = 30
def _probe_mantle(client: ManagementClient) -> ConnectionTestResponse:
return unwrap(
client.connection_test(
ConnectionTestBody(
litellm_params=LiteLLMParamsBody(
model=MANTLE_RESPONSES_BACKEND, aws_region_name=MANTLE_REGION
),
mode="responses",
)
)
)
class TestModelTestConnection:
@pytest.mark.covers("mgmt.model.test_connection.happy_path")
def test_bedrock_mantle_responses_connection_succeeds(self, client: ManagementClient) -> None:
response = unwrap(
client.connection_test(
ConnectionTestBody(
litellm_params=LiteLLMParamsBody(
model=MANTLE_RESPONSES_BACKEND, aws_region_name=MANTLE_REGION
),
mode="responses",
for attempt in range(1, PROBE_ATTEMPTS + 1):
response = _probe_mantle(client)
if response.status == "success":
return
error = response.result.error if response.result else None
assert error == PROBE_TIMEOUT_ERROR, f"test_connection reported an error: {error}"
if attempt < PROBE_ATTEMPTS:
print(
f"test_connection probe timed out; retry {attempt}/{PROBE_ATTEMPTS - 1}"
f" in {PROBE_RETRY_SLEEP_SECONDS}s",
flush=True,
)
)
)
error = response.result.error if response.result else None
assert response.status == "success", f"test_connection reported an error: {error}"
time.sleep(PROBE_RETRY_SLEEP_SECONDS)
pytest.fail(f"test_connection timed out on all {PROBE_ATTEMPTS} attempts")