diff --git a/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py b/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py index 1bc3623e025..224784bd1dc 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py +++ b/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py @@ -93,7 +93,7 @@ _BEDROCK_TOO_LARGE_ERROR_SUBSTRINGS = ( "too large", "exceeds the maximum", ) -_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS = 20_000 +_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS = 25_000 _BEDROCK_APPLY_GUARDRAIL_MAX_THROTTLE_RETRIES = 3 _BEDROCK_APPLY_GUARDRAIL_BASE_BACKOFF_SECONDS = 0.5 # Resource-less, detect-only InvokeGuardrailChecks API (no guardrail resource required). @@ -216,12 +216,14 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM): content_filter_threshold: float | None = 0.5, prompt_attack_threshold: float | None = 0.5, pii_confidence_threshold: float | None = 0.5, + chunk_budget_chars: int = _BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS, **kwargs, ): self.async_handler = get_async_httpx_client(llm_provider=httpxSpecialProvider.GuardrailCallback) self.guardrailIdentifier = guardrailIdentifier self.guardrailVersion = guardrailVersion self.guardrail_provider = "bedrock" + self.chunk_budget_chars = chunk_budget_chars self.experimental_use_latest_role_message_only = bool(kwargs.get("experimental_use_latest_role_message_only")) # Resource-less, detect-only InvokeGuardrailChecks mode. Present `checks` @@ -847,9 +849,7 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM): content: list[BedrockContentItem] = bedrock_request_data.get("content") or [] allow_chunking = not self._content_uses_contextual_grounding(content) batches = ( - self._bin_pack_bedrock_content(content, budget=_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS) - if allow_chunking - else [content] + self._bin_pack_bedrock_content(content, budget=self.chunk_budget_chars) if allow_chunking else [content] ) try: @@ -1199,14 +1199,19 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM): too large, `_apply_guardrail_content_with_chunking`'s existing recursive-bisection fallback takes over for that batch only. - `budget` (`_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS` at the call site) - is a conservative starting guess, not a correctness dependency. It only - sets how many calls the common case takes. AWS's real per-request - text-unit cap depends on account, region, and policy, is not a fixed - character count, and cannot be read from config, so any batch it still - rejects falls back to bisection, which self-corrects however wrong the - guess was. A too-generous guess therefore costs one extra probe-and-bisect - round trip, the same one pure reactive bisection would have paid anyway. + `budget` comes from the guardrail's ``chunk_budget_chars`` setting and + defaults to 25,000, matching ApplyGuardrail's default quota of 25 text + units (roughly 1,000 characters each) per second. Packing to that size and + posting sequentially is what keeps chunking from tripping the rate quota + and trading a size error for a throttle. Accounts with raised quotas can + configure a larger budget to spend fewer calls. + + The budget is not a correctness dependency either way. AWS's effective cap + varies by account, region, and policy, is not a fixed character count, and + cannot be read from config, so any batch it still rejects falls back to + bisection, which self-corrects however wrong the value was. An over-large + budget therefore costs one extra probe-and-bisect round trip rather than + failing the request. """ if not content: return [content] diff --git a/litellm/types/guardrails.py b/litellm/types/guardrails.py index 5b611971154..bc03e26bcfa 100644 --- a/litellm/types/guardrails.py +++ b/litellm/types/guardrails.py @@ -526,6 +526,17 @@ class BedrockGuardrailConfigModel(BaseModel): description="InvokeGuardrailChecks: block when any sensitiveInformation confidenceScore " ">= this value (scores are in [0,1]). Set to null to make PII detection detect-only.", ) + chunk_budget_chars: int = Field( + default=25_000, + gt=0, + description="ApplyGuardrail: how much content (in characters) to send per call. " + "Content above this is split across several sequential calls. Defaults to 25,000, " + "matching ApplyGuardrail's default quota of 25 text units (~1,000 characters each) " + "per second, so chunking does not trip that rate limit. Raise it if your account's " + "quotas have been increased; a batch AWS still rejects as too large is bisected " + "automatically, so an over-large value costs an extra round trip rather than " + "failing the request.", + ) class LakeraV2GuardrailConfigModel(BaseModel): diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py index 3580b95ebaa..d5f580266c7 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py @@ -4325,3 +4325,54 @@ async def test_apply_guardrail_too_large_reported_as_429_bisects_without_burning assert result.get("action") == "NONE" output_texts = [o.get("text") for o in result.get("outputs") or []] assert output_texts == ["half-2", "half-3"] + + +def test_chunk_budget_defaults_to_apply_guardrail_per_second_quota(): + """The default budget must track ApplyGuardrail's default quota of 25 text units + (about 1,000 characters each) per second. Packing to that size and posting + sequentially is what stops chunking from trading a size error for a throttle, so + this default is a deliberate match to AWS behaviour rather than an arbitrary + number.""" + assert _BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS == 25_000 + assert BedrockGuardrail(guardrailIdentifier="g", guardrailVersion="DRAFT").chunk_budget_chars == 25_000 + + +@pytest.mark.asyncio +async def test_configured_chunk_budget_changes_how_content_is_packed(): + """An account with raised quotas can set a larger `chunk_budget_chars` and have it + actually drive packing, spending fewer ApplyGuardrail calls for the same content + instead of being pinned to the conservative default. + + Four 20,000-character messages are 80,000 characters total. At the 25,000 default + only one message fits per batch, so it takes four calls; at 100,000 all four fit + in a single batch, so it takes one.""" + messages = [{"role": "user", "content": "x" * 20_000} for _ in range(4)] + + mock_credentials = MagicMock() + mock_credentials.access_key = "k" + mock_credentials.secret_key = "s" + mock_credentials.token = None + + async def _calls_made_with_budget(budget: int) -> int: + guardrail = BedrockGuardrail( + guardrail_name="test-bedrock-guard", + guardrailIdentifier="test-guardrail", + guardrailVersion="DRAFT", + disable_exception_on_block=False, + chunk_budget_chars=budget, + ) + with ( + patch.object(guardrail.async_handler, "post", new_callable=AsyncMock) as mock_post, + patch.object(guardrail, "_load_credentials", return_value=(mock_credentials, "us-east-1")), + patch.object(guardrail, "_prepare_request", return_value=MagicMock()), + ): + mock_post.side_effect = lambda *_a, **_k: _passing_bedrock_httpx_response("ok") + await guardrail.make_bedrock_api_request( + source="INPUT", + messages=messages, + request_data={"model": "bedrock-nova-micro"}, + ) + return mock_post.await_count + + assert await _calls_made_with_budget(_BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS) == 4 + assert await _calls_made_with_budget(100_000) == 1