From 6b8e988ff02a6b943467acc754e35b29871a663b Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 21 Sep 2026 13:30:25 -0700 Subject: [PATCH] test(e2e): settle for the replica propagation window and trim the disconnect cell's prose --- tests/e2e/e2e_http.py | 11 ++--- tests/e2e/router/reliability_support.py | 7 +-- ...st_reliability_cancel_on_disconnect_e2e.py | 47 ++++++++----------- .../router/test_reliability_cooldowns_e2e.py | 2 +- 4 files changed, 27 insertions(+), 40 deletions(-) diff --git a/tests/e2e/e2e_http.py b/tests/e2e/e2e_http.py index 58817d399a8..97f1e1671f8 100644 --- a/tests/e2e/e2e_http.py +++ b/tests/e2e/e2e_http.py @@ -677,9 +677,8 @@ def send( class AbandonedRequest(BaseModel): - """A non-streaming request the client walked away from: the socket was closed - ``after`` seconds in, before the proxy had answered, so the proxy saw a client - disconnect with the upstream call still in flight.""" + """A non-streaming request whose socket the client closed ``after`` seconds in, + before the proxy had answered.""" kind: Literal["abandoned"] = "abandoned" after: float @@ -688,10 +687,8 @@ class AbandonedRequest(BaseModel): def abandon( url: URL, *, headers: BaseModel, json: BaseModel, after: float, connect_timeout: float = 10.0 ) -> AbandonedRequest | StreamingResponse: - """POST and hang up ``after`` seconds if no response head has arrived by then, - closing the connection so the proxy observes the disconnect. Returns the - response instead when the proxy answered first, so a test can tell a real - disconnect from a generation that finished too fast to be cancelled.""" + """POST and close the connection ``after`` seconds if no response head has arrived + by then; returns the response instead when the proxy answered first.""" sent_at: Final = time.monotonic() session: Final = requests.Session() try: diff --git a/tests/e2e/router/reliability_support.py b/tests/e2e/router/reliability_support.py index 887954bc7fd..3d5b76f6408 100644 --- a/tests/e2e/router/reliability_support.py +++ b/tests/e2e/router/reliability_support.py @@ -53,6 +53,7 @@ CONTENT_POLICY_PROMPT = ( ) COOLDOWN_SECONDS = 30.0 +REPLICA_PROPAGATION_SECONDS = 15.0 # The smallest-context chat model OpenAI still serves (16385 tokens). A prompt # past that limit comes back as a real `context_length_exceeded` 400, which is @@ -122,11 +123,7 @@ def create_content_filtered_deployment(proxy: ProxyClient, name: str) -> str: def create_azure_benched_on_first_failure_deployment(proxy: ProxyClient, name: str, cooldown_time: float) -> str: """The live Azure OpenAI deployment holding all of the group's shuffle weight, - benched on its first failure of any class, with the client's own retries off. - The 500 the proxy used to book against a call the client hung up on carries no - provider body, so litellm maps it to a bare APIError that no named - allowed_fails_policy class covers; the deployment-wide allowed_fails=0 is the - knob that makes that undeserved bench show on the very next call.""" + benched on its first failure of any class, with the client's own retries off.""" return proxy.register_model( ModelNewBody( model_name=name, diff --git a/tests/e2e/router/test_reliability_cancel_on_disconnect_e2e.py b/tests/e2e/router/test_reliability_cancel_on_disconnect_e2e.py index 7d46a980034..110b540057c 100644 --- a/tests/e2e/router/test_reliability_cancel_on_disconnect_e2e.py +++ b/tests/e2e/router/test_reliability_cancel_on_disconnect_e2e.py @@ -1,27 +1,19 @@ """Live e2e: a client hanging up mid-request under cancel_on_disconnect never benches the deployment it was talking to. -With `general_settings.cancel_on_disconnect: true` the proxy cancels the in-flight -provider call the moment the client's socket closes. The Azure handler used to -turn that cancellation into a fake 500, which the router booked as a deployment -failure: one impatient client benched a healthy deployment and every caller -behind it paid for fallbacks (GitHub issues #35329 and #42222). This cell pins the -fix at the seam a customer sees. The group is the cooldown suite's pair: the live -Azure deployment holding all of the shuffle weight, benched on its first failure -of any class (the fake 500 carried no provider body, so litellm mapped it to a -bare APIError no named policy class covers) with a cooldown long enough to -outlast the test, plus a healthy backup at weight 0 the shuffle can only reach -once the Azure deployment is benched. One cheap call first proves the Azure -deployment answers the key and leaves the key's auth path warm. The test then -asks for a long answer, retries off, and hangs up a few seconds in: the client's -read timeout closes the socket well after the proxy has handed the call to Azure -(a cold virtual-key auth can take a couple of seconds on its own, and a hang-up -that lands before the provider call is in flight cancels nothing the router could -bench, so a shorter window passes vacuously) and well before the answer is done. -After a settle window wide enough for a sibling replica to have read any bench -from Redis, every one of the next calls has to come back 200 from the Azure -deployment itself, named in x-litellm-model-id; a single answer from the backup -means the hang-up was booked as a failure. +The group is the cooldown suite's pair: the live Azure deployment holding all of +the shuffle weight, benched on its first failure of any class with a cooldown that +outlasts the test, plus a healthy backup at weight 0 the shuffle only reaches once +the Azure deployment is benched. A cheap call first proves the Azure deployment +answers the key and warms its auth path. The test then asks for an answer far +longer than CLIENT_HANGS_UP_AFTER_SECONDS of generation, retries off, and hangs up +that many seconds in: late enough that the proxy has handed the call to Azure (a +hang-up before the provider call is in flight cancels nothing the router could +bench, so the cell would pass vacuously), and should the proxy ever answer first +the cell fails out loud naming the window instead of passing. After the cooldown +suite's replica propagation window, every one of the next calls has to come back +200 from the Azure deployment itself, named in x-litellm-model-id; a single answer +from the backup means the hang-up was booked as a failure. The test reads `cancel_on_disconnect` back from the proxy first: without the flag the hang-up cancels nothing and the cell would pass vacuously. @@ -38,6 +30,7 @@ from e2e_http import AbandonedRequest, StreamingResponse from lifecycle import ResourceManager from models import ChatMessage, ReliabilityChatBody, RouterSettingsOverride from reliability_support import ( + REPLICA_PROPAGATION_SECONDS, chat_override, create_azure_benched_on_first_failure_deployment, create_zero_weight_backup_deployment, @@ -47,9 +40,8 @@ from reliability_support import ( pytestmark = pytest.mark.e2e CLIENT_HANGS_UP_AFTER_SECONDS = 8.0 -LONG_ANSWER_MAX_TOKENS = 4096 +LONG_ANSWER_MAX_TOKENS = 16384 BENCH_OUTLASTS_TEST_SECONDS = 300.0 -SETTLE_AFTER_HANGUP_SECONDS = 3.0 CALLS_AFTER_HANGUP = 6 @@ -64,8 +56,6 @@ def _say_hi(client: ComplexityRouterClient, key: str, group: str) -> StreamingRe def _hang_up_mid_answer(client: ComplexityRouterClient, key: str, group: str) -> None: - """Send a request whose answer takes far longer than the client waits, so the - client closes the socket while the provider is still generating.""" outcome = client.proxy.transport.abandon( "/chat/completions", headers=client.proxy.transport.bearer(key), @@ -74,7 +64,10 @@ def _hang_up_mid_answer(client: ComplexityRouterClient, key: str, group: str) -> messages=[ ChatMessage( role="user", - content=f"Write a 3000 word essay on the history of the telegraph. {unique_marker()}", + content=( + "Write a 10000 word essay on the history of the telegraph, one section per decade. " + f"{unique_marker()}" + ), ) ], max_tokens=LONG_ANSWER_MAX_TOKENS, @@ -117,7 +110,7 @@ class TestReliabilityCancelOnDisconnect: ) _hang_up_mid_answer(client, scoped_key, group) - time.sleep(SETTLE_AFTER_HANGUP_SECONDS) + time.sleep(REPLICA_PROPAGATION_SECONDS) for call in range(1, CALLS_AFTER_HANGUP + 1): resp = _say_hi(client, scoped_key, group) diff --git a/tests/e2e/router/test_reliability_cooldowns_e2e.py b/tests/e2e/router/test_reliability_cooldowns_e2e.py index 5b5cec09f06..769971e1533 100644 --- a/tests/e2e/router/test_reliability_cooldowns_e2e.py +++ b/tests/e2e/router/test_reliability_cooldowns_e2e.py @@ -43,6 +43,7 @@ from lifecycle import ResourceManager from models import KeyGenerateBody, RouterSettingsOverride from reliability_support import ( COOLDOWN_SECONDS, + REPLICA_PROPAGATION_SECONDS, chat_override, create_always_5xx_deployment, create_always_rate_limited_deployment, @@ -57,7 +58,6 @@ from reliability_support import ( pytestmark = pytest.mark.e2e RECOVERY_GRACE_SECONDS = 10 -REPLICA_PROPAGATION_SECONDS = 15.0 PROPAGATION_POLL_SECONDS = 0.25 BENCH_MARGIN_SECONDS = 4.0