mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(guardrails): don't retry an oversized Bedrock guardrail call as a throttle
AWS reports an ApplyGuardrail request that exceeds the per-request text-unit cap as a ThrottlingException (429), not only as the documented ValidationException (400). Verified against a live guardrail with an active content-filter policy: a 3273-text-unit request comes back as "Input text size (3273 text units) exceeds the maximum allowed (1000 text units) for the content filter policy (Classic tier)". The throttle retry keyed off status 429 alone, so every oversized chunk burned the full backoff-retry budget - each attempt a billed AWS call preceded by a sleep - before the bisection fallback got a chance, at every level of the recursion. A size error is not transient; re-posting the same content can never succeed. It now short-circuits straight to bisection. Also rename _is_input_too_large_validation_error to _is_input_too_large_error (it never keyed off the status code, and the error is not always a ValidationException), correct the docstrings that asserted a 400, and log at warning level when a split happens so the recovery is visible without --detailed_debug.
This commit is contained in:
parent
61dfe02c80
commit
10c7493494
2 changed files with 112 additions and 7 deletions
|
|
@ -914,8 +914,10 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
|
||||
Tries `content` as a single call first. AWS's per-request "maximum input
|
||||
size in text units" quota is account/region/policy-dependent and cannot be
|
||||
predicted ahead of time, so it is only ever discovered reactively: on a 400
|
||||
ValidationException whose message indicates the input was too large, the
|
||||
predicted ahead of time, so it is only ever discovered reactively: on an
|
||||
error whose message indicates the input was too large (a ThrottlingException
|
||||
in practice, a ValidationException per the docs -- see
|
||||
``_is_input_too_large_error``), the
|
||||
content is split in half and each half is retried the same way (recursing
|
||||
until every piece fits or cannot be split further). A single oversized
|
||||
content item (one very long message) is split by its own text instead of
|
||||
|
|
@ -946,12 +948,19 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
)
|
||||
]
|
||||
except HTTPException as exc:
|
||||
if allow_chunking and self._is_input_too_large_validation_error(exc.detail):
|
||||
if allow_chunking and self._is_input_too_large_error(exc.detail):
|
||||
split_content = self._split_bedrock_content(content)
|
||||
if split_content is None:
|
||||
raise
|
||||
first_half, second_half = split_content
|
||||
is_text_fragment = len(content) == 1
|
||||
verbose_proxy_logger.warning(
|
||||
"Bedrock Guardrail: ApplyGuardrail rejected %d content item(s) as too large; "
|
||||
"splitting into %d + %d and retrying each",
|
||||
len(content),
|
||||
len(first_half),
|
||||
len(second_half),
|
||||
)
|
||||
first_results = await self._apply_guardrail_content_with_chunking(
|
||||
content=first_half,
|
||||
base_request_data=base_request_data,
|
||||
|
|
@ -998,6 +1007,13 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
retries here are capped low -- they must not multiply per-request latency
|
||||
by an order of magnitude when the account's per-second text-unit quota is
|
||||
the binding constraint rather than the per-request size quota.
|
||||
|
||||
A too-large rejection is deliberately excluded from the retry. AWS reports
|
||||
it as a ThrottlingException (429), not only as a ValidationException, but
|
||||
unlike a genuine throttle it is not transient: re-posting the same
|
||||
oversized content can never succeed. Retrying it would burn every backoff
|
||||
sleep and every (billed) attempt before the caller's bisection gets a
|
||||
chance to split the content, at every level of the recursion.
|
||||
"""
|
||||
attempt = 0
|
||||
while True:
|
||||
|
|
@ -1013,7 +1029,11 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
start_time=start_time,
|
||||
)
|
||||
except HTTPException as exc:
|
||||
if exc.status_code == 429 and attempt < _BEDROCK_APPLY_GUARDRAIL_MAX_THROTTLE_RETRIES:
|
||||
if (
|
||||
exc.status_code == 429
|
||||
and not self._is_input_too_large_error(exc.detail)
|
||||
and attempt < _BEDROCK_APPLY_GUARDRAIL_MAX_THROTTLE_RETRIES
|
||||
):
|
||||
await asyncio.sleep(_BEDROCK_APPLY_GUARDRAIL_BASE_BACKOFF_SECONDS * (2**attempt))
|
||||
attempt += 1
|
||||
continue
|
||||
|
|
@ -1274,9 +1294,17 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
return left + 1 if midpoint - left <= right - midpoint else right + 1
|
||||
|
||||
@staticmethod
|
||||
def _is_input_too_large_validation_error(detail: object) -> bool:
|
||||
"""True if `detail` is the AWS ValidationException message for input
|
||||
exceeding the per-request text-unit quota.
|
||||
def _is_input_too_large_error(detail: object) -> bool:
|
||||
"""True if `detail` is an AWS error message for input exceeding the
|
||||
per-request text-unit quota.
|
||||
|
||||
Matched on the message rather than the status code on purpose: AWS is not
|
||||
consistent about which error it raises for this. Observed against a live
|
||||
guardrail with an active content-filter policy, an oversized request comes
|
||||
back as a *ThrottlingException* (429) reading ``Input text size (3273 text
|
||||
units) exceeds the maximum allowed (1000 text units) for the content filter
|
||||
policy (Classic tier)``, while the documented failure mode is a
|
||||
ValidationException (400). Keying off the message covers both.
|
||||
|
||||
A guardrail *block* is also raised as an HTTPException with status 400,
|
||||
but its ``detail`` is always a dict (built by
|
||||
|
|
|
|||
|
|
@ -3321,6 +3321,23 @@ def _throttling_httpx_response() -> MagicMock:
|
|||
return response
|
||||
|
||||
|
||||
def _too_large_throttling_httpx_response() -> MagicMock:
|
||||
"""The shape AWS actually returns for an oversized ApplyGuardrail request when
|
||||
the guardrail has an active content-filter policy: a 429 ThrottlingException,
|
||||
not the documented 400 ValidationException. Message taken from a live call."""
|
||||
response = MagicMock()
|
||||
response.status_code = 429
|
||||
response.json.return_value = {
|
||||
"message": (
|
||||
"Input text size (3273 text units) exceeds the maximum allowed "
|
||||
"(1000 text units) for the content filter policy (Classic tier)."
|
||||
)
|
||||
}
|
||||
response.text = json.dumps(response.json.return_value)
|
||||
response.headers = {}
|
||||
return response
|
||||
|
||||
|
||||
def _passing_bedrock_httpx_response(marker: str) -> MagicMock:
|
||||
"""A successful ApplyGuardrail response tagged with `marker` so tests can
|
||||
verify which chunk produced which output/usage after merging."""
|
||||
|
|
@ -4124,3 +4141,63 @@ def test_bin_pack_bedrock_content_empty_content_makes_exactly_one_empty_batch():
|
|||
pre-bin-packing behavior of sending the content list as-is in one call --
|
||||
bin-packing must not turn an empty request into zero ApplyGuardrail calls."""
|
||||
assert BedrockGuardrail._bin_pack_bedrock_content([], budget=100) == [[]]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_apply_guardrail_too_large_reported_as_429_bisects_without_burning_retries():
|
||||
"""AWS reports an oversized ApplyGuardrail request as a 429 ThrottlingException
|
||||
(not the documented 400 ValidationException) when the guardrail has an active
|
||||
content-filter policy. That is not a transient throttle -- re-posting the same
|
||||
oversized content can never succeed -- so it must bisect immediately instead of
|
||||
consuming the exponential-backoff retry budget first.
|
||||
|
||||
Regression for a bug found against a live guardrail: because the throttle retry
|
||||
only keyed off status 429, every oversized chunk burned all
|
||||
_BEDROCK_APPLY_GUARDRAIL_MAX_THROTTLE_RETRIES attempts (each a billed AWS call,
|
||||
each preceded by a backoff sleep) before bisection got a chance, at every level
|
||||
of the recursion."""
|
||||
guardrail = _bedrock_guardrail_for_chunk_tests()
|
||||
|
||||
messages = [
|
||||
{"role": "user", "content": "chunk one text"},
|
||||
{"role": "user", "content": "chunk two text"},
|
||||
]
|
||||
|
||||
mock_credentials = MagicMock()
|
||||
mock_credentials.access_key = "k"
|
||||
mock_credentials.secret_key = "s"
|
||||
mock_credentials.token = None
|
||||
|
||||
call_count = 0
|
||||
|
||||
async def _post_side_effect(*_args, **_kwargs):
|
||||
nonlocal call_count
|
||||
call_count += 1
|
||||
if call_count == 1:
|
||||
return _too_large_throttling_httpx_response()
|
||||
return _passing_bedrock_httpx_response(f"half-{call_count}")
|
||||
|
||||
with (
|
||||
patch.object(guardrail.async_handler, "post", new_callable=AsyncMock) as mock_post,
|
||||
patch.object(guardrail, "_load_credentials", return_value=(mock_credentials, "us-east-1")),
|
||||
patch.object(guardrail, "_prepare_request", return_value=MagicMock()),
|
||||
patch(
|
||||
"litellm.proxy.guardrails.guardrail_hooks.bedrock_guardrails.asyncio.sleep",
|
||||
new_callable=AsyncMock,
|
||||
) as mock_sleep,
|
||||
):
|
||||
mock_post.side_effect = _post_side_effect
|
||||
|
||||
result = await guardrail.make_bedrock_api_request(
|
||||
source="INPUT",
|
||||
messages=messages,
|
||||
request_data={"model": "bedrock-nova-micro"},
|
||||
)
|
||||
|
||||
# One rejected whole-content call, then exactly one call per bisected half.
|
||||
assert call_count == 3
|
||||
# No backoff sleep: a size error must not be treated as a transient throttle.
|
||||
mock_sleep.assert_not_awaited()
|
||||
assert result.get("action") == "NONE"
|
||||
output_texts = [o.get("text") for o in result.get("outputs") or []]
|
||||
assert output_texts == ["half-2", "half-3"]
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue