mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
feat(guardrails): match AWS default chunk budget and make it configurable
ApplyGuardrail's default quota is 25 text units, roughly 25,000 characters, per second. Chunking has to respect that throughput limit rather than just the per-request size, otherwise splitting an oversized request trades a size error for a throttle. The budget now defaults to 25,000 to match that default for every user, up from an arbitrary 20,000. Accounts with raised quotas can spend fewer calls by setting chunk_budget_chars on the guardrail. A value AWS still rejects as too large is bisected automatically, so an over-large setting costs an extra round trip rather than failing the request.
This commit is contained in:
parent
c64a02c4a3
commit
ffe529c04c
3 changed files with 79 additions and 12 deletions
|
|
@ -93,7 +93,7 @@ _BEDROCK_TOO_LARGE_ERROR_SUBSTRINGS = (
|
|||
"too large",
|
||||
"exceeds the maximum",
|
||||
)
|
||||
_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS = 20_000
|
||||
_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS = 25_000
|
||||
_BEDROCK_APPLY_GUARDRAIL_MAX_THROTTLE_RETRIES = 3
|
||||
_BEDROCK_APPLY_GUARDRAIL_BASE_BACKOFF_SECONDS = 0.5
|
||||
# Resource-less, detect-only InvokeGuardrailChecks API (no guardrail resource required).
|
||||
|
|
@ -216,12 +216,14 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
content_filter_threshold: float | None = 0.5,
|
||||
prompt_attack_threshold: float | None = 0.5,
|
||||
pii_confidence_threshold: float | None = 0.5,
|
||||
chunk_budget_chars: int = _BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS,
|
||||
**kwargs,
|
||||
):
|
||||
self.async_handler = get_async_httpx_client(llm_provider=httpxSpecialProvider.GuardrailCallback)
|
||||
self.guardrailIdentifier = guardrailIdentifier
|
||||
self.guardrailVersion = guardrailVersion
|
||||
self.guardrail_provider = "bedrock"
|
||||
self.chunk_budget_chars = chunk_budget_chars
|
||||
self.experimental_use_latest_role_message_only = bool(kwargs.get("experimental_use_latest_role_message_only"))
|
||||
|
||||
# Resource-less, detect-only InvokeGuardrailChecks mode. Present `checks`
|
||||
|
|
@ -847,9 +849,7 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
content: list[BedrockContentItem] = bedrock_request_data.get("content") or []
|
||||
allow_chunking = not self._content_uses_contextual_grounding(content)
|
||||
batches = (
|
||||
self._bin_pack_bedrock_content(content, budget=_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS)
|
||||
if allow_chunking
|
||||
else [content]
|
||||
self._bin_pack_bedrock_content(content, budget=self.chunk_budget_chars) if allow_chunking else [content]
|
||||
)
|
||||
|
||||
try:
|
||||
|
|
@ -1199,14 +1199,19 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
too large, `_apply_guardrail_content_with_chunking`'s existing
|
||||
recursive-bisection fallback takes over for that batch only.
|
||||
|
||||
`budget` (`_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS` at the call site)
|
||||
is a conservative starting guess, not a correctness dependency. It only
|
||||
sets how many calls the common case takes. AWS's real per-request
|
||||
text-unit cap depends on account, region, and policy, is not a fixed
|
||||
character count, and cannot be read from config, so any batch it still
|
||||
rejects falls back to bisection, which self-corrects however wrong the
|
||||
guess was. A too-generous guess therefore costs one extra probe-and-bisect
|
||||
round trip, the same one pure reactive bisection would have paid anyway.
|
||||
`budget` comes from the guardrail's ``chunk_budget_chars`` setting and
|
||||
defaults to 25,000, matching ApplyGuardrail's default quota of 25 text
|
||||
units (roughly 1,000 characters each) per second. Packing to that size and
|
||||
posting sequentially is what keeps chunking from tripping the rate quota
|
||||
and trading a size error for a throttle. Accounts with raised quotas can
|
||||
configure a larger budget to spend fewer calls.
|
||||
|
||||
The budget is not a correctness dependency either way. AWS's effective cap
|
||||
varies by account, region, and policy, is not a fixed character count, and
|
||||
cannot be read from config, so any batch it still rejects falls back to
|
||||
bisection, which self-corrects however wrong the value was. An over-large
|
||||
budget therefore costs one extra probe-and-bisect round trip rather than
|
||||
failing the request.
|
||||
"""
|
||||
if not content:
|
||||
return [content]
|
||||
|
|
|
|||
|
|
@ -526,6 +526,17 @@ class BedrockGuardrailConfigModel(BaseModel):
|
|||
description="InvokeGuardrailChecks: block when any sensitiveInformation confidenceScore "
|
||||
">= this value (scores are in [0,1]). Set to null to make PII detection detect-only.",
|
||||
)
|
||||
chunk_budget_chars: int = Field(
|
||||
default=25_000,
|
||||
gt=0,
|
||||
description="ApplyGuardrail: how much content (in characters) to send per call. "
|
||||
"Content above this is split across several sequential calls. Defaults to 25,000, "
|
||||
"matching ApplyGuardrail's default quota of 25 text units (~1,000 characters each) "
|
||||
"per second, so chunking does not trip that rate limit. Raise it if your account's "
|
||||
"quotas have been increased; a batch AWS still rejects as too large is bisected "
|
||||
"automatically, so an over-large value costs an extra round trip rather than "
|
||||
"failing the request.",
|
||||
)
|
||||
|
||||
|
||||
class LakeraV2GuardrailConfigModel(BaseModel):
|
||||
|
|
|
|||
|
|
@ -4325,3 +4325,54 @@ async def test_apply_guardrail_too_large_reported_as_429_bisects_without_burning
|
|||
assert result.get("action") == "NONE"
|
||||
output_texts = [o.get("text") for o in result.get("outputs") or []]
|
||||
assert output_texts == ["half-2", "half-3"]
|
||||
|
||||
|
||||
def test_chunk_budget_defaults_to_apply_guardrail_per_second_quota():
|
||||
"""The default budget must track ApplyGuardrail's default quota of 25 text units
|
||||
(about 1,000 characters each) per second. Packing to that size and posting
|
||||
sequentially is what stops chunking from trading a size error for a throttle, so
|
||||
this default is a deliberate match to AWS behaviour rather than an arbitrary
|
||||
number."""
|
||||
assert _BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS == 25_000
|
||||
assert BedrockGuardrail(guardrailIdentifier="g", guardrailVersion="DRAFT").chunk_budget_chars == 25_000
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_configured_chunk_budget_changes_how_content_is_packed():
|
||||
"""An account with raised quotas can set a larger `chunk_budget_chars` and have it
|
||||
actually drive packing, spending fewer ApplyGuardrail calls for the same content
|
||||
instead of being pinned to the conservative default.
|
||||
|
||||
Four 20,000-character messages are 80,000 characters total. At the 25,000 default
|
||||
only one message fits per batch, so it takes four calls; at 100,000 all four fit
|
||||
in a single batch, so it takes one."""
|
||||
messages = [{"role": "user", "content": "x" * 20_000} for _ in range(4)]
|
||||
|
||||
mock_credentials = MagicMock()
|
||||
mock_credentials.access_key = "k"
|
||||
mock_credentials.secret_key = "s"
|
||||
mock_credentials.token = None
|
||||
|
||||
async def _calls_made_with_budget(budget: int) -> int:
|
||||
guardrail = BedrockGuardrail(
|
||||
guardrail_name="test-bedrock-guard",
|
||||
guardrailIdentifier="test-guardrail",
|
||||
guardrailVersion="DRAFT",
|
||||
disable_exception_on_block=False,
|
||||
chunk_budget_chars=budget,
|
||||
)
|
||||
with (
|
||||
patch.object(guardrail.async_handler, "post", new_callable=AsyncMock) as mock_post,
|
||||
patch.object(guardrail, "_load_credentials", return_value=(mock_credentials, "us-east-1")),
|
||||
patch.object(guardrail, "_prepare_request", return_value=MagicMock()),
|
||||
):
|
||||
mock_post.side_effect = lambda *_a, **_k: _passing_bedrock_httpx_response("ok")
|
||||
await guardrail.make_bedrock_api_request(
|
||||
source="INPUT",
|
||||
messages=messages,
|
||||
request_data={"model": "bedrock-nova-micro"},
|
||||
)
|
||||
return mock_post.await_count
|
||||
|
||||
assert await _calls_made_with_budget(_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS) == 4
|
||||
assert await _calls_made_with_budget(100_000) == 1
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue