fix(retry): treat provider usage-limit errors as terminal, not transient

A plan/subscription usage-limit rejection (e.g. ChatGPT's 5-hour or weekly cap,
HTTP 429 `usage_limit_reached`) was classified as an ordinary transient throttle
and retried by both retry layers — the SDK ModelSettings retry
(`DEFAULT_MODEL_RETRY`, up to 5x) and Strix's turn replay
(`_is_transient_model_error` -> `_run_cycle`, up to 5x). Retrying cannot succeed
until the window resets (often hours away); it only burns more of the same
exhausted quota and spends minutes thrashing exponential backoff, flooding the
log with `openai.agents: Error streaming response` before the run finally stops.

Classify usage-limit errors as terminal so the run fails fast and lands on the
existing resumable stop path (`run_strix_scan` catches the RateLimitError, logs
the `strix --resume <run>` hint, and stops the root):

- add `codex.is_usage_limit_error()` (text-marker predicate, mirroring
  `is_content_guardrail_error`);
- exclude it from `_is_transient_model_error` (Strix replay layer);
- exclude it from the SDK retry policy via a small wrapper
  (`_default_model_retry_policy`).

Ordinary 429 throttles are unchanged and still retried. Adds regression tests.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
Samir Hanna Verza 2026-08-26 09:33:45 -03:00
parent bfaaa904f2
commit e7c311acb9
5 changed files with 63 additions and 5 deletions

View file

@ -160,6 +160,21 @@ def is_content_guardrail_error(exc: BaseException) -> bool:
return any(marker in text for marker in _GUARDRAIL_MARKERS)
_USAGE_LIMIT_MARKERS = (
"usage_limit_reached",
"usage limit has been reached",
)
def is_usage_limit_error(exc: BaseException) -> bool:
"""A plan/subscription usage window is exhausted (e.g. ChatGPT's 5-hour or
weekly cap), not an ordinary throttle. Terminal for the run: retrying only
burns more of the same exhausted quota and cannot succeed until it resets,
so the run should stop resumably instead of thrashing the backoff."""
text = str(exc).lower()
return any(marker in text for marker in _USAGE_LIMIT_MARKERS)
def _b64url(raw: bytes) -> str:
return base64.urlsafe_b64encode(raw).rstrip(b"=").decode("ascii")

View file

@ -496,6 +496,15 @@ class StrixProvider(MultiProvider):
)
def _not_usage_limit_error(context: RetryPolicyContext) -> bool:
# A usage-limit rejection (a plan/subscription window is exhausted, e.g.
# ChatGPT's 5-hour or weekly cap) is terminal for the run: retrying only
# burns more of the same exhausted quota and cannot succeed until it resets.
# Composed with ``all`` below so the run fails fast and stops resumably
# instead of thrashing the backoff on a limit that will not clear for hours.
return not codex.is_usage_limit_error(context.error)
DEFAULT_MODEL_RETRY = ModelRetrySettings(
max_retries=5,
backoff=ModelRetryBackoffSettings(
@ -504,11 +513,14 @@ DEFAULT_MODEL_RETRY = ModelRetrySettings(
multiplier=2.0,
jitter=False,
),
policy=retry_policies.any(
retry_policies.provider_suggested(),
retry_policies.network_error(),
retry_policies.http_status((429, 500, 502, 503, 504)),
_retry_statusless_provider_errors,
policy=retry_policies.all(
retry_policies.any(
retry_policies.provider_suggested(),
retry_policies.network_error(),
retry_policies.http_status((429, 500, 502, 503, 504)),
_retry_statusless_provider_errors,
),
_not_usage_limit_error,
),
)

View file

@ -132,6 +132,8 @@ def _model_error_status_code(exc: BaseException) -> int | None:
def _is_transient_model_error(exc: BaseException) -> bool:
if codex.is_content_guardrail_error(exc):
return False
if codex.is_usage_limit_error(exc):
return False
if isinstance(
exc, APITimeoutError | APIConnectionError | TimeoutError | ConnectionError | OSError
):

View file

@ -172,3 +172,23 @@ async def test_run_cycle_does_not_retry_permanent_error(
streams = [_FakeStream(exc=bad_request), _FakeStream()]
with pytest.raises(BadRequestError):
await _run_once(monkeypatch, streams)
def test_usage_limit_error_is_terminal_not_transient() -> None:
usage_limited = RateLimitError(
"Error code: 429 - {'error': {'type': 'usage_limit_reached', "
"'message': 'The usage limit has been reached'}}",
response=httpx.Response(429, request=_request()),
body={"type": "usage_limit_reached"},
)
assert codex.is_usage_limit_error(usage_limited) is True
# A usage-limit rejection must not be retried like an ordinary throttle.
assert execution._is_transient_model_error(usage_limited) is False
def test_ordinary_rate_limit_is_not_usage_limit() -> None:
rate_limited = RateLimitError(
"slow down", response=httpx.Response(429, request=_request()), body=None
)
assert codex.is_usage_limit_error(rate_limited) is False
assert execution._is_transient_model_error(rate_limited) is True

View file

@ -88,3 +88,12 @@ def test_policy_helper_matches_statusless_only() -> None:
)
is False
)
def test_usage_limit_error_is_not_retried() -> None:
# A usage-limit rejection carries a 429, but the window will not clear for
# hours; retrying only burns more of the same quota, so it must be terminal
# even though an ordinary 429 is retried.
usage_limited = RuntimeError("Error code: 429 - {'error': {'type': 'usage_limit_reached'}}")
assert codex.is_usage_limit_error(usage_limited) is True
assert _retries(ModelRetryNormalizedError(status_code=429), error=usage_limited) is False