feat(guardrails): match AWS default chunk budget and make it configurable

ApplyGuardrail's default quota is 25 text units, roughly 25,000 characters,
per second. Chunking has to respect that throughput limit rather than just the
per-request size, otherwise splitting an oversized request trades a size error
for a throttle. The budget now defaults to 25,000 to match that default for
every user, up from an arbitrary 20,000.

Accounts with raised quotas can spend fewer calls by setting
chunk_budget_chars on the guardrail. A value AWS still rejects as too large is
bisected automatically, so an over-large setting costs an extra round trip
rather than failing the request.
This commit is contained in:
spencer-burridge 2026-07-29 14:14:42 -05:00
parent c64a02c4a3
commit ffe529c04c
3 changed files with 79 additions and 12 deletions

View file

@ -93,7 +93,7 @@ _BEDROCK_TOO_LARGE_ERROR_SUBSTRINGS = (
"too large",
"exceeds the maximum",
)
_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS = 20_000
_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS = 25_000
_BEDROCK_APPLY_GUARDRAIL_MAX_THROTTLE_RETRIES = 3
_BEDROCK_APPLY_GUARDRAIL_BASE_BACKOFF_SECONDS = 0.5
# Resource-less, detect-only InvokeGuardrailChecks API (no guardrail resource required).
@ -216,12 +216,14 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
content_filter_threshold: float | None = 0.5,
prompt_attack_threshold: float | None = 0.5,
pii_confidence_threshold: float | None = 0.5,
chunk_budget_chars: int = _BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS,
**kwargs,
):
self.async_handler = get_async_httpx_client(llm_provider=httpxSpecialProvider.GuardrailCallback)
self.guardrailIdentifier = guardrailIdentifier
self.guardrailVersion = guardrailVersion
self.guardrail_provider = "bedrock"
self.chunk_budget_chars = chunk_budget_chars
self.experimental_use_latest_role_message_only = bool(kwargs.get("experimental_use_latest_role_message_only"))
# Resource-less, detect-only InvokeGuardrailChecks mode. Present `checks`
@ -847,9 +849,7 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
content: list[BedrockContentItem] = bedrock_request_data.get("content") or []
allow_chunking = not self._content_uses_contextual_grounding(content)
batches = (
self._bin_pack_bedrock_content(content, budget=_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS)
if allow_chunking
else [content]
self._bin_pack_bedrock_content(content, budget=self.chunk_budget_chars) if allow_chunking else [content]
)
try:
@ -1199,14 +1199,19 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
too large, `_apply_guardrail_content_with_chunking`'s existing
recursive-bisection fallback takes over for that batch only.
`budget` (`_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS` at the call site)
is a conservative starting guess, not a correctness dependency. It only
sets how many calls the common case takes. AWS's real per-request
text-unit cap depends on account, region, and policy, is not a fixed
character count, and cannot be read from config, so any batch it still
rejects falls back to bisection, which self-corrects however wrong the
guess was. A too-generous guess therefore costs one extra probe-and-bisect
round trip, the same one pure reactive bisection would have paid anyway.
`budget` comes from the guardrail's ``chunk_budget_chars`` setting and
defaults to 25,000, matching ApplyGuardrail's default quota of 25 text
units (roughly 1,000 characters each) per second. Packing to that size and
posting sequentially is what keeps chunking from tripping the rate quota
and trading a size error for a throttle. Accounts with raised quotas can
configure a larger budget to spend fewer calls.
The budget is not a correctness dependency either way. AWS's effective cap
varies by account, region, and policy, is not a fixed character count, and
cannot be read from config, so any batch it still rejects falls back to
bisection, which self-corrects however wrong the value was. An over-large
budget therefore costs one extra probe-and-bisect round trip rather than
failing the request.
"""
if not content:
return [content]

View file

@ -526,6 +526,17 @@ class BedrockGuardrailConfigModel(BaseModel):
description="InvokeGuardrailChecks: block when any sensitiveInformation confidenceScore "
">= this value (scores are in [0,1]). Set to null to make PII detection detect-only.",
)
chunk_budget_chars: int = Field(
default=25_000,
gt=0,
description="ApplyGuardrail: how much content (in characters) to send per call. "
"Content above this is split across several sequential calls. Defaults to 25,000, "
"matching ApplyGuardrail's default quota of 25 text units (~1,000 characters each) "
"per second, so chunking does not trip that rate limit. Raise it if your account's "
"quotas have been increased; a batch AWS still rejects as too large is bisected "
"automatically, so an over-large value costs an extra round trip rather than "
"failing the request.",
)
class LakeraV2GuardrailConfigModel(BaseModel):

View file

@ -4325,3 +4325,54 @@ async def test_apply_guardrail_too_large_reported_as_429_bisects_without_burning
assert result.get("action") == "NONE"
output_texts = [o.get("text") for o in result.get("outputs") or []]
assert output_texts == ["half-2", "half-3"]
def test_chunk_budget_defaults_to_apply_guardrail_per_second_quota():
"""The default budget must track ApplyGuardrail's default quota of 25 text units
(about 1,000 characters each) per second. Packing to that size and posting
sequentially is what stops chunking from trading a size error for a throttle, so
this default is a deliberate match to AWS behaviour rather than an arbitrary
number."""
assert _BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS == 25_000
assert BedrockGuardrail(guardrailIdentifier="g", guardrailVersion="DRAFT").chunk_budget_chars == 25_000
@pytest.mark.asyncio
async def test_configured_chunk_budget_changes_how_content_is_packed():
"""An account with raised quotas can set a larger `chunk_budget_chars` and have it
actually drive packing, spending fewer ApplyGuardrail calls for the same content
instead of being pinned to the conservative default.
Four 20,000-character messages are 80,000 characters total. At the 25,000 default
only one message fits per batch, so it takes four calls; at 100,000 all four fit
in a single batch, so it takes one."""
messages = [{"role": "user", "content": "x" * 20_000} for _ in range(4)]
mock_credentials = MagicMock()
mock_credentials.access_key = "k"
mock_credentials.secret_key = "s"
mock_credentials.token = None
async def _calls_made_with_budget(budget: int) -> int:
guardrail = BedrockGuardrail(
guardrail_name="test-bedrock-guard",
guardrailIdentifier="test-guardrail",
guardrailVersion="DRAFT",
disable_exception_on_block=False,
chunk_budget_chars=budget,
)
with (
patch.object(guardrail.async_handler, "post", new_callable=AsyncMock) as mock_post,
patch.object(guardrail, "_load_credentials", return_value=(mock_credentials, "us-east-1")),
patch.object(guardrail, "_prepare_request", return_value=MagicMock()),
):
mock_post.side_effect = lambda *_a, **_k: _passing_bedrock_httpx_response("ok")
await guardrail.make_bedrock_api_request(
source="INPUT",
messages=messages,
request_data={"model": "bedrock-nova-micro"},
)
return mock_post.await_count
assert await _calls_made_with_budget(_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS) == 4
assert await _calls_made_with_budget(100_000) == 1