diff --git a/strix/config/codex.py b/strix/config/codex.py index cf34f003..1f87e600 100644 --- a/strix/config/codex.py +++ b/strix/config/codex.py @@ -160,6 +160,21 @@ def is_content_guardrail_error(exc: BaseException) -> bool: return any(marker in text for marker in _GUARDRAIL_MARKERS) +_USAGE_LIMIT_MARKERS = ( + "usage_limit_reached", + "usage limit has been reached", +) + + +def is_usage_limit_error(exc: BaseException) -> bool: + """A plan/subscription usage window is exhausted (e.g. ChatGPT's 5-hour or + weekly cap), not an ordinary throttle. Terminal for the run: retrying only + burns more of the same exhausted quota and cannot succeed until it resets, + so the run should stop resumably instead of thrashing the backoff.""" + text = str(exc).lower() + return any(marker in text for marker in _USAGE_LIMIT_MARKERS) + + def _b64url(raw: bytes) -> str: return base64.urlsafe_b64encode(raw).rstrip(b"=").decode("ascii") diff --git a/strix/config/models.py b/strix/config/models.py index f6848ca4..a7b5d51f 100644 --- a/strix/config/models.py +++ b/strix/config/models.py @@ -496,6 +496,15 @@ class StrixProvider(MultiProvider): ) +def _not_usage_limit_error(context: RetryPolicyContext) -> bool: + # A usage-limit rejection (a plan/subscription window is exhausted, e.g. + # ChatGPT's 5-hour or weekly cap) is terminal for the run: retrying only + # burns more of the same exhausted quota and cannot succeed until it resets. + # Composed with ``all`` below so the run fails fast and stops resumably + # instead of thrashing the backoff on a limit that will not clear for hours. + return not codex.is_usage_limit_error(context.error) + + DEFAULT_MODEL_RETRY = ModelRetrySettings( max_retries=5, backoff=ModelRetryBackoffSettings( @@ -504,11 +513,14 @@ DEFAULT_MODEL_RETRY = ModelRetrySettings( multiplier=2.0, jitter=False, ), - policy=retry_policies.any( - retry_policies.provider_suggested(), - retry_policies.network_error(), - retry_policies.http_status((429, 500, 502, 503, 504)), - _retry_statusless_provider_errors, + policy=retry_policies.all( + retry_policies.any( + retry_policies.provider_suggested(), + retry_policies.network_error(), + retry_policies.http_status((429, 500, 502, 503, 504)), + _retry_statusless_provider_errors, + ), + _not_usage_limit_error, ), ) diff --git a/strix/core/execution.py b/strix/core/execution.py index 9f2674d2..b61cd205 100644 --- a/strix/core/execution.py +++ b/strix/core/execution.py @@ -132,6 +132,8 @@ def _model_error_status_code(exc: BaseException) -> int | None: def _is_transient_model_error(exc: BaseException) -> bool: if codex.is_content_guardrail_error(exc): return False + if codex.is_usage_limit_error(exc): + return False if isinstance( exc, APITimeoutError | APIConnectionError | TimeoutError | ConnectionError | OSError ): diff --git a/tests/test_execution_transient_retry.py b/tests/test_execution_transient_retry.py index d81e4e70..c64472fb 100644 --- a/tests/test_execution_transient_retry.py +++ b/tests/test_execution_transient_retry.py @@ -172,3 +172,23 @@ async def test_run_cycle_does_not_retry_permanent_error( streams = [_FakeStream(exc=bad_request), _FakeStream()] with pytest.raises(BadRequestError): await _run_once(monkeypatch, streams) + + +def test_usage_limit_error_is_terminal_not_transient() -> None: + usage_limited = RateLimitError( + "Error code: 429 - {'error': {'type': 'usage_limit_reached', " + "'message': 'The usage limit has been reached'}}", + response=httpx.Response(429, request=_request()), + body={"type": "usage_limit_reached"}, + ) + assert codex.is_usage_limit_error(usage_limited) is True + # A usage-limit rejection must not be retried like an ordinary throttle. + assert execution._is_transient_model_error(usage_limited) is False + + +def test_ordinary_rate_limit_is_not_usage_limit() -> None: + rate_limited = RateLimitError( + "slow down", response=httpx.Response(429, request=_request()), body=None + ) + assert codex.is_usage_limit_error(rate_limited) is False + assert execution._is_transient_model_error(rate_limited) is True diff --git a/tests/test_model_retry.py b/tests/test_model_retry.py index 4cbffbcd..60ffe69d 100644 --- a/tests/test_model_retry.py +++ b/tests/test_model_retry.py @@ -88,3 +88,12 @@ def test_policy_helper_matches_statusless_only() -> None: ) is False ) + + +def test_usage_limit_error_is_not_retried() -> None: + # A usage-limit rejection carries a 429, but the window will not clear for + # hours; retrying only burns more of the same quota, so it must be terminal + # even though an ordinary 429 is retried. + usage_limited = RuntimeError("Error code: 429 - {'error': {'type': 'usage_limit_reached'}}") + assert codex.is_usage_limit_error(usage_limited) is True + assert _retries(ModelRetryNormalizedError(status_code=429), error=usage_limited) is False