mirror of
https://github.com/usestrix/strix.git
synced 2026-10-08 03:08:08 +00:00
feat(llm): retry 4xx model errors (#1486)
* feat(llm): log the upstream provider behind OpenRouter LiteLLM drops OpenRouter's top-level `provider` from stream chunks (including error chunks) and the `metadata.provider_name` of non-2xx replies, so the request log could not say which provider served or failed an attempt. Record both, plus OpenRouter's `metadata.error_type`, per attempt and emit them as `upstream=` / `upstream_error=` on the llm_request line. * feat(llm): retry 4xx model errors * test: let the blocked-provider test run out of 4xx retries * refactor: retry any 4xx or 5xx status * fix: don't retry 401, 402 or 403 * fix: don't retry 404
This commit is contained in:
parent
39792c9d61
commit
928f053610
2 changed files with 7 additions and 11 deletions
|
|
@ -184,9 +184,7 @@ def _is_transient_model_error(exc: BaseException) -> bool:
|
|||
return True
|
||||
code = _model_error_status_code(exc)
|
||||
if code is not None:
|
||||
import litellm
|
||||
|
||||
return bool(litellm._should_retry(code))
|
||||
return code >= 400 and code not in (401, 402, 403, 404)
|
||||
return isinstance(exc, APIError)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -79,12 +79,13 @@ def test_content_guardrail_is_not_retried() -> None:
|
|||
assert execution._is_transient_model_error(guardrail) is False
|
||||
|
||||
|
||||
def test_client_errors_are_not_transient() -> None:
|
||||
def test_client_errors_are_transient() -> None:
|
||||
bad_request = BadRequestError(
|
||||
"bad", response=httpx.Response(400, request=_request()), body=None
|
||||
)
|
||||
assert execution._is_transient_model_error(bad_request) is False
|
||||
assert execution._is_transient_model_error(_status_error(404)) is False
|
||||
assert execution._is_transient_model_error(bad_request) is True
|
||||
for status in (401, 402, 403, 404):
|
||||
assert execution._is_transient_model_error(_status_error(status)) is False
|
||||
assert execution._is_transient_model_error(ValueError("nope")) is False
|
||||
|
||||
|
||||
|
|
@ -167,11 +168,8 @@ async def test_run_cycle_gives_up_after_max_retries(
|
|||
async def test_run_cycle_does_not_retry_permanent_error(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
bad_request = BadRequestError(
|
||||
"bad", response=httpx.Response(400, request=_request()), body=None
|
||||
)
|
||||
streams = [_FakeStream(exc=bad_request), _FakeStream()]
|
||||
with pytest.raises(BadRequestError):
|
||||
streams = [_FakeStream(exc=ValueError("nope")), _FakeStream()]
|
||||
with pytest.raises(ValueError, match="nope"):
|
||||
await _run_once(monkeypatch, streams)
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue