From a6ad663af1e8c62cdd7cbff734dd0aff4d54033e Mon Sep 17 00:00:00 2001 From: Yucheng Zhu Date: Thu, 6 Aug 2026 13:49:55 -0700 Subject: [PATCH] style(guardrails): move chunking rationale out of comments and into docstrings --- litellm/constants.py | 2 -- .../guardrail_hooks/bedrock_guardrails.py | 23 ++++++++++--------- .../test_bedrock_guardrails.py | 2 -- 3 files changed, 12 insertions(+), 15 deletions(-) diff --git a/litellm/constants.py b/litellm/constants.py index fe26275f571..d8a81c22727 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -279,8 +279,6 @@ TOOL_POLICY_CACHE_TTL_SECONDS: Final = int(os.getenv("TOOL_POLICY_CACHE_TTL_SECO GUARDRAIL_SCANNED_MESSAGES_CACHE_TTL_SECONDS: Final = int( os.getenv("GUARDRAIL_SCANNED_MESSAGES_CACHE_TTL_SECONDS", 24 * 60 * 60) ) -# Batch size Bedrock ApplyGuardrail content is packed into once AWS has rejected a -# payload as too large. Shared by BedrockGuardrailConfigModel's default and the guardrail BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS: Final = 25_000 # Aggregation threshold: default to 80% of the asyncio queue maxsize so the check can always trigger. # Must be < LITELLM_ASYNCIO_QUEUE_MAXSIZE; if set higher the aggregation logic will never fire. diff --git a/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py b/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py index 7e3b07b1d95..37ae82b0a1e 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py +++ b/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py @@ -1076,6 +1076,11 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM): parse the result. Raises HTTPException on a guardrail block or any non-200 response (including 429, handled by the retry wrapper above). + AWS also reports some failures inside a 200 body, tagging ``Output.__type`` + with an Exception marker, and those raise here too so the one consolidated + log entry for the request records ``guardrail_failed_to_respond`` instead of + a success. + A block is logged here rather than by the caller: it ends the whole chunking flow immediately, with no further chunks attempted, so there is no later merged response for the caller to log instead. @@ -1104,9 +1109,6 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM): ) if httpx_response.status_code == 200: - # AWS can report a failure inside a 200 body (Output.__type carries an - # Exception marker). Raise so the one consolidated log entry for this - # request records guardrail_failed_to_respond rather than success if self._check_bedrock_response_for_exception(httpx_response): status_code, detail_message = self._parse_bedrock_guardrail_error_response(httpx_response) raise HTTPException(status_code=status_code, detail=detail_message) @@ -1224,7 +1226,8 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM): budget: int, ) -> list[list[BedrockContentItem]]: """Pack whole content items, in order, into batches whose combined text - length stays within `budget`. + length stays within `budget`, in a single pass that carries the running + total rather than re-summing the open batch per item. This is the fast-path half of the hybrid chunking strategy: bin-packing at a conservative fixed budget keeps the common case at O(n / budget) @@ -1260,8 +1263,6 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM): return batch_index, used + length return batch_index + 1, length - # One pass: give each item a batch number, then group runs of equal numbers. - # Carrying the running total avoids re-summing the open batch per item batch_numbers: Final = (index for index, _ in tuple(accumulate(lengths, assign, initial=(0, 0)))[1:]) return [ [item for _, item in group] for _, group in groupby(zip(batch_numbers, content), key=lambda pair: pair[0]) @@ -1395,6 +1396,11 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM): -- mirroring a real single-call response and matching what ``_build_tracing_detail`` treats as "Bedrock didn't report an action". + Fields this merge has no opinion on (``actionReason``, ``guardrailCoverage``, + ``blockedResponse``, anything AWS adds later) are carried over from the chunk + responses rather than dropped, so the response and the logged telemetry keep + the shape a single unchunked call returned. The merged keys below win. + Per AWS's documented ApplyGuardrail contract, a single call's ``outputs`` is positionally parallel to the ``content`` items *of that call*: an entry per item when anything in the call was masked, or an empty list @@ -1434,10 +1440,6 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM): ] any_usage_reported = any(chunk_result.response.get("usage") for chunk_result in chunk_results) - # Seed from the raw chunk responses so fields this merge has no opinion on - # (actionReason, guardrailCoverage, blockedResponse, anything AWS adds later) - # survive instead of being dropped, which is what a single unchunked call - # returned before chunking existed. The merged keys below then win merged: Final[BedrockGuardrailResponse] = cast( BedrockGuardrailResponse, {key: value for chunk_result in chunk_results for key, value in chunk_result.response.items()}, @@ -1445,7 +1447,6 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM): if merged_action is not None: merged["action"] = merged_action if merged_outputs and any_masked: - # AWS reports both spellings; keep them consistent so every reader agrees merged["outputs"] = merged_outputs merged["output"] = merged_outputs if merged_assessments: diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py index 2cf63c18b6a..819a6dc20c8 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py @@ -3728,7 +3728,6 @@ async def test_apply_guardrail_accepted_content_costs_exactly_one_call(): latency on traffic that never had a size problem.""" guardrail = _bedrock_guardrail_for_chunk_tests() - # 3 x half-budget items is 1.5x the budget, so eager packing would send 2 calls item_text = "x" * (BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS // 2) messages = [{"role": "user", "content": item_text} for _ in range(3)] @@ -3825,7 +3824,6 @@ async def test_apply_guardrail_batch_under_budget_still_rejected_falls_back_to_b nonlocal call_count call_count += 1 if call_count in (1, 2): - # 1 = whole-payload probe, 2 = the first packed batch, both over the real cap return _too_large_validation_httpx_response() return _passing_bedrock_httpx_response(f"chunk-{call_count}")