style(guardrails): move chunking rationale out of comments and into docstrings

This commit is contained in:
Yucheng Zhu 2026-08-06 13:49:55 -07:00
parent 07662ac153
commit a6ad663af1
3 changed files with 12 additions and 15 deletions

View file

@ -279,8 +279,6 @@ TOOL_POLICY_CACHE_TTL_SECONDS: Final = int(os.getenv("TOOL_POLICY_CACHE_TTL_SECO
GUARDRAIL_SCANNED_MESSAGES_CACHE_TTL_SECONDS: Final = int(
os.getenv("GUARDRAIL_SCANNED_MESSAGES_CACHE_TTL_SECONDS", 24 * 60 * 60)
)
# Batch size Bedrock ApplyGuardrail content is packed into once AWS has rejected a
# payload as too large. Shared by BedrockGuardrailConfigModel's default and the guardrail
BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS: Final = 25_000
# Aggregation threshold: default to 80% of the asyncio queue maxsize so the check can always trigger.
# Must be < LITELLM_ASYNCIO_QUEUE_MAXSIZE; if set higher the aggregation logic will never fire.

View file

@ -1076,6 +1076,11 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
parse the result. Raises HTTPException on a guardrail block or any
non-200 response (including 429, handled by the retry wrapper above).
AWS also reports some failures inside a 200 body, tagging ``Output.__type``
with an Exception marker, and those raise here too so the one consolidated
log entry for the request records ``guardrail_failed_to_respond`` instead of
a success.
A block is logged here rather than by the caller: it ends the whole chunking
flow immediately, with no further chunks attempted, so there is no later
merged response for the caller to log instead.
@ -1104,9 +1109,6 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
)
if httpx_response.status_code == 200:
# AWS can report a failure inside a 200 body (Output.__type carries an
# Exception marker). Raise so the one consolidated log entry for this
# request records guardrail_failed_to_respond rather than success
if self._check_bedrock_response_for_exception(httpx_response):
status_code, detail_message = self._parse_bedrock_guardrail_error_response(httpx_response)
raise HTTPException(status_code=status_code, detail=detail_message)
@ -1224,7 +1226,8 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
budget: int,
) -> list[list[BedrockContentItem]]:
"""Pack whole content items, in order, into batches whose combined text
length stays within `budget`.
length stays within `budget`, in a single pass that carries the running
total rather than re-summing the open batch per item.
This is the fast-path half of the hybrid chunking strategy: bin-packing
at a conservative fixed budget keeps the common case at O(n / budget)
@ -1260,8 +1263,6 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
return batch_index, used + length
return batch_index + 1, length
# One pass: give each item a batch number, then group runs of equal numbers.
# Carrying the running total avoids re-summing the open batch per item
batch_numbers: Final = (index for index, _ in tuple(accumulate(lengths, assign, initial=(0, 0)))[1:])
return [
[item for _, item in group] for _, group in groupby(zip(batch_numbers, content), key=lambda pair: pair[0])
@ -1395,6 +1396,11 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
-- mirroring a real single-call response and matching what
``_build_tracing_detail`` treats as "Bedrock didn't report an action".
Fields this merge has no opinion on (``actionReason``, ``guardrailCoverage``,
``blockedResponse``, anything AWS adds later) are carried over from the chunk
responses rather than dropped, so the response and the logged telemetry keep
the shape a single unchunked call returned. The merged keys below win.
Per AWS's documented ApplyGuardrail contract, a single call's ``outputs``
is positionally parallel to the ``content`` items *of that call*: an
entry per item when anything in the call was masked, or an empty list
@ -1434,10 +1440,6 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
]
any_usage_reported = any(chunk_result.response.get("usage") for chunk_result in chunk_results)
# Seed from the raw chunk responses so fields this merge has no opinion on
# (actionReason, guardrailCoverage, blockedResponse, anything AWS adds later)
# survive instead of being dropped, which is what a single unchunked call
# returned before chunking existed. The merged keys below then win
merged: Final[BedrockGuardrailResponse] = cast(
BedrockGuardrailResponse,
{key: value for chunk_result in chunk_results for key, value in chunk_result.response.items()},
@ -1445,7 +1447,6 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
if merged_action is not None:
merged["action"] = merged_action
if merged_outputs and any_masked:
# AWS reports both spellings; keep them consistent so every reader agrees
merged["outputs"] = merged_outputs
merged["output"] = merged_outputs
if merged_assessments:

View file

@ -3728,7 +3728,6 @@ async def test_apply_guardrail_accepted_content_costs_exactly_one_call():
latency on traffic that never had a size problem."""
guardrail = _bedrock_guardrail_for_chunk_tests()
# 3 x half-budget items is 1.5x the budget, so eager packing would send 2 calls
item_text = "x" * (BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS // 2)
messages = [{"role": "user", "content": item_text} for _ in range(3)]
@ -3825,7 +3824,6 @@ async def test_apply_guardrail_batch_under_budget_still_rejected_falls_back_to_b
nonlocal call_count
call_count += 1
if call_count in (1, 2):
# 1 = whole-payload probe, 2 = the first packed batch, both over the real cap
return _too_large_validation_httpx_response()
return _passing_bedrock_httpx_response(f"chunk-{call_count}")