mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
style(guardrails): move chunking rationale out of comments and into docstrings
This commit is contained in:
parent
07662ac153
commit
a6ad663af1
3 changed files with 12 additions and 15 deletions
|
|
@ -279,8 +279,6 @@ TOOL_POLICY_CACHE_TTL_SECONDS: Final = int(os.getenv("TOOL_POLICY_CACHE_TTL_SECO
|
|||
GUARDRAIL_SCANNED_MESSAGES_CACHE_TTL_SECONDS: Final = int(
|
||||
os.getenv("GUARDRAIL_SCANNED_MESSAGES_CACHE_TTL_SECONDS", 24 * 60 * 60)
|
||||
)
|
||||
# Batch size Bedrock ApplyGuardrail content is packed into once AWS has rejected a
|
||||
# payload as too large. Shared by BedrockGuardrailConfigModel's default and the guardrail
|
||||
BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS: Final = 25_000
|
||||
# Aggregation threshold: default to 80% of the asyncio queue maxsize so the check can always trigger.
|
||||
# Must be < LITELLM_ASYNCIO_QUEUE_MAXSIZE; if set higher the aggregation logic will never fire.
|
||||
|
|
|
|||
|
|
@ -1076,6 +1076,11 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
parse the result. Raises HTTPException on a guardrail block or any
|
||||
non-200 response (including 429, handled by the retry wrapper above).
|
||||
|
||||
AWS also reports some failures inside a 200 body, tagging ``Output.__type``
|
||||
with an Exception marker, and those raise here too so the one consolidated
|
||||
log entry for the request records ``guardrail_failed_to_respond`` instead of
|
||||
a success.
|
||||
|
||||
A block is logged here rather than by the caller: it ends the whole chunking
|
||||
flow immediately, with no further chunks attempted, so there is no later
|
||||
merged response for the caller to log instead.
|
||||
|
|
@ -1104,9 +1109,6 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
)
|
||||
|
||||
if httpx_response.status_code == 200:
|
||||
# AWS can report a failure inside a 200 body (Output.__type carries an
|
||||
# Exception marker). Raise so the one consolidated log entry for this
|
||||
# request records guardrail_failed_to_respond rather than success
|
||||
if self._check_bedrock_response_for_exception(httpx_response):
|
||||
status_code, detail_message = self._parse_bedrock_guardrail_error_response(httpx_response)
|
||||
raise HTTPException(status_code=status_code, detail=detail_message)
|
||||
|
|
@ -1224,7 +1226,8 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
budget: int,
|
||||
) -> list[list[BedrockContentItem]]:
|
||||
"""Pack whole content items, in order, into batches whose combined text
|
||||
length stays within `budget`.
|
||||
length stays within `budget`, in a single pass that carries the running
|
||||
total rather than re-summing the open batch per item.
|
||||
|
||||
This is the fast-path half of the hybrid chunking strategy: bin-packing
|
||||
at a conservative fixed budget keeps the common case at O(n / budget)
|
||||
|
|
@ -1260,8 +1263,6 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
return batch_index, used + length
|
||||
return batch_index + 1, length
|
||||
|
||||
# One pass: give each item a batch number, then group runs of equal numbers.
|
||||
# Carrying the running total avoids re-summing the open batch per item
|
||||
batch_numbers: Final = (index for index, _ in tuple(accumulate(lengths, assign, initial=(0, 0)))[1:])
|
||||
return [
|
||||
[item for _, item in group] for _, group in groupby(zip(batch_numbers, content), key=lambda pair: pair[0])
|
||||
|
|
@ -1395,6 +1396,11 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
-- mirroring a real single-call response and matching what
|
||||
``_build_tracing_detail`` treats as "Bedrock didn't report an action".
|
||||
|
||||
Fields this merge has no opinion on (``actionReason``, ``guardrailCoverage``,
|
||||
``blockedResponse``, anything AWS adds later) are carried over from the chunk
|
||||
responses rather than dropped, so the response and the logged telemetry keep
|
||||
the shape a single unchunked call returned. The merged keys below win.
|
||||
|
||||
Per AWS's documented ApplyGuardrail contract, a single call's ``outputs``
|
||||
is positionally parallel to the ``content`` items *of that call*: an
|
||||
entry per item when anything in the call was masked, or an empty list
|
||||
|
|
@ -1434,10 +1440,6 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
]
|
||||
any_usage_reported = any(chunk_result.response.get("usage") for chunk_result in chunk_results)
|
||||
|
||||
# Seed from the raw chunk responses so fields this merge has no opinion on
|
||||
# (actionReason, guardrailCoverage, blockedResponse, anything AWS adds later)
|
||||
# survive instead of being dropped, which is what a single unchunked call
|
||||
# returned before chunking existed. The merged keys below then win
|
||||
merged: Final[BedrockGuardrailResponse] = cast(
|
||||
BedrockGuardrailResponse,
|
||||
{key: value for chunk_result in chunk_results for key, value in chunk_result.response.items()},
|
||||
|
|
@ -1445,7 +1447,6 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
if merged_action is not None:
|
||||
merged["action"] = merged_action
|
||||
if merged_outputs and any_masked:
|
||||
# AWS reports both spellings; keep them consistent so every reader agrees
|
||||
merged["outputs"] = merged_outputs
|
||||
merged["output"] = merged_outputs
|
||||
if merged_assessments:
|
||||
|
|
|
|||
|
|
@ -3728,7 +3728,6 @@ async def test_apply_guardrail_accepted_content_costs_exactly_one_call():
|
|||
latency on traffic that never had a size problem."""
|
||||
guardrail = _bedrock_guardrail_for_chunk_tests()
|
||||
|
||||
# 3 x half-budget items is 1.5x the budget, so eager packing would send 2 calls
|
||||
item_text = "x" * (BEDROCK_APPLY_GUARDRAIL_CHUNK_BUDGET_CHARS // 2)
|
||||
messages = [{"role": "user", "content": item_text} for _ in range(3)]
|
||||
|
||||
|
|
@ -3825,7 +3824,6 @@ async def test_apply_guardrail_batch_under_budget_still_rejected_falls_back_to_b
|
|||
nonlocal call_count
|
||||
call_count += 1
|
||||
if call_count in (1, 2):
|
||||
# 1 = whole-payload probe, 2 = the first packed batch, both over the real cap
|
||||
return _too_large_validation_httpx_response()
|
||||
return _passing_bedrock_httpx_response(f"chunk-{call_count}")
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue