From ba7df45027f980116504344f2662419c7265f1ce Mon Sep 17 00:00:00 2001 From: Sean Turner Date: Mon, 27 Apr 2026 22:50:11 +0100 Subject: [PATCH] fix: decouple transient-thinking retry budget from outer max_retries MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Greptile caught a real logic bug in the initial patch: the `continue` inside the transient-thinking branch advanced the outer `for attempt in range(max_retries + 1)` counter, so each transient retry consumed a generic-retry slot. With the default `max_retries=5`: - Loop runs 6 iterations (attempts 0–5) - Each transient retry burned one iteration - On attempt 5, `attempt >= max_retries` fired `_raise_error` before the 6th sleep (160s) ever ran - The documented "fall through to generic retry" path was unreachable Fix: inner `while` loop that does its own `_stream()` retry without advancing the outer `attempt` counter. The transient budget of 6 (5 / 10 / 20 / 40 / 80 / 160 s) is now independent of `max_retries`, and once the inner budget is exhausted the most recent exception falls through to the generic retry path as the PR originally intended. Also extracts the literal `6` into `max_transient_thinking_retries` for readability — still unconfigurable since the envelope budget is based on observed Bedrock transient durations, not something we want users tuning blindly. --- strix/llm/llm.py | 22 ++++++++++++++++------ 1 file changed, 16 insertions(+), 6 deletions(-) diff --git a/strix/llm/llm.py b/strix/llm/llm.py index 8db71fa3..448aca6f 100644 --- a/strix/llm/llm.py +++ b/strix/llm/llm.py @@ -160,6 +160,7 @@ class LLM: max_retries = int(Config.get("strix_llm_max_retries") or "5") transient_thinking_retries = 0 + max_transient_thinking_retries = 6 for attempt in range(max_retries + 1): try: @@ -171,14 +172,23 @@ class LLM: # blocks have been modified — even when the payload is structurally # valid and the same payload succeeds on replay. Observed on # claude-sonnet-4-6 with adaptive thinking. Retry with exponential - # backoff (5s, 10s, 20s, 40s, 80s, 160s — ~5 min total budget) before - # falling through to the generic retry path. - if self._is_transient_thinking_error(e) and transient_thinking_retries < 6: + # backoff (5s, 10s, 20s, 40s, 80s, 160s — ~5 min total budget), using + # an inner loop so the transient retry counter does not share slots + # with the outer max_retries budget. Once the inner budget is + # exhausted (or a non-transient error is raised), fall through to the + # generic retry path below. + while ( + self._is_transient_thinking_error(e) + and transient_thinking_retries < max_transient_thinking_retries + ): transient_thinking_retries += 1 - if attempt >= max_retries: - self._raise_error(e) await asyncio.sleep(min(240, 5 * (2 ** (transient_thinking_retries - 1)))) - continue + try: + async for response in self._stream(messages): + yield response + return # noqa: TRY300 + except Exception as e2: # noqa: BLE001 + e = e2 if attempt >= max_retries or not self._should_retry(e): self._raise_error(e) wait = min(90, 2 * (2**attempt))