litellm/tests/claude_code/pdf_input/test_bedrock_invoke.py
mateo-berri d05e45893e compat-matrix: parallel-fanout refactor + 5 new feature dirs + rate limiter
- Switch all per-cell tests from @pytest.mark.parametrize("model", ...)
  (3 sequential invocations) to a single test that fans out to all 3
  Claude tiers via run_claude_models_parallel. Per-cell wall time is now
  bounded by the slowest model rather than the sum.

- Add 5 new v0 feature dirs (5 providers each, 25 new test files):
    web_search, pdf_input, prompt_caching_1h,
    tool_use_streaming, thinking_with_tool_use
  Manifest expanded to match.

- Add cross-process token-bucket rate limiter (rate_limiter.py + tests)
  so xdist workers stay under per-provider req/s limits during full-grid
  runs. New env knobs: LITELLM_COMPAT_RATE_{ANTHROPIC,AZURE,VERTEX_AI,
  BEDROCK_CONVERSE,BEDROCK_INVOKE}.

- conftest.py: write per-worker shards under <artifact>.shards/, merge
  in the controller; preserve the "don't write empty artifact" guard so
  unit-test runs don't clobber a real compat-results.json.

- Vertex test_config.yaml: route project/location through env so the
  cron VM can target a different GCP project than the upstream default.

- Add run_compat.sh wrapper for binary-searching ideal req/s per
  provider against compat-rate-limit-summary.json output.
2026-05-06 23:31:19 +00:00

151 lines
4.8 KiB
Python

"""pdf_input x Bedrock (Invoke).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
write a tiny valid PDF to disk, allow the built-in `Read` tool, and
ask Claude to read the PDF and report what it contains.
Bedrock InvokeModel for Anthropic models accepts the native
`document` content block shape; this cell catches gateway regressions
where the proxy drops or mis-encodes the document content block on
the way through (e.g. base64-only encoding, missing media_type, etc.).
The (feature, provider) for this cell is inferred from the file path by
`tests/claude_code/conftest.py`:
tests/claude_code/pdf_input/test_bedrock_invoke.py
^^^^^^^^^ ^^^^^^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
import pytest
from tests.claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_INVOKE_MODELS = [
"claude-haiku-4-5-bedrock-invoke",
"claude-sonnet-4-6-bedrock-invoke",
"claude-opus-4-7-bedrock-invoke",
]
PDF_MARKER = "PONG"
def _build_minimal_pdf(marker: str) -> bytes:
"""Return a single-page PDF whose only visible text is `marker`."""
objects = [
b"<< /Type /Catalog /Pages 2 0 R >>",
b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
(
b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] "
b"/Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>"
),
(
b"<< /Length %d >>\nstream\nBT /F1 24 Tf 50 100 Td ("
+ marker.encode("ascii")
+ b") Tj ET\nendstream"
),
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
]
body_open = objects[3].index(b"\nstream\n") + len(b"\nstream\n")
body_close = objects[3].index(b"\nendstream")
body_len = body_close - body_open
objects[3] = (
b"<< /Length "
+ str(body_len).encode("ascii")
+ b" >>\nstream\n"
+ objects[3][body_open:body_close]
+ b"\nendstream"
)
out = bytearray(b"%PDF-1.4\n")
offsets = []
for i, obj in enumerate(objects, start=1):
offsets.append(len(out))
out += f"{i} 0 obj\n".encode("ascii") + obj + b"\nendobj\n"
xref_offset = len(out)
out += b"xref\n0 %d\n" % (len(objects) + 1)
out += b"0000000000 65535 f \n"
for off in offsets:
out += f"{off:010d} 00000 n \n".encode("ascii")
out += (
b"trailer\n<< /Size "
+ str(len(objects) + 1).encode("ascii")
+ b" /Root 1 0 R >>\nstartxref\n"
+ str(xref_offset).encode("ascii")
+ b"\n%%EOF\n"
)
return bytes(out)
def test_pdf_input_bedrock_invoke(compat_result, tmp_path):
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
pdf_path = tmp_path / "marker.pdf"
pdf_path.write_bytes(_build_minimal_pdf(PDF_MARKER))
outcomes = run_claude_models_parallel(
models=BEDROCK_INVOKE_MODELS,
prompt=(
f"Use the Read tool to read the file at {pdf_path}. "
"Report the single word that appears in the document."
),
base_url=base_url,
api_key=api_key,
extra_args=["--allowed-tools", "Read"],
)
failures = []
for model in BEDROCK_INVOKE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if PDF_MARKER not in outcome.text.upper():
error = (
f"[{model}] reply did not reference the PDF marker {PDF_MARKER!r}; "
f"got: {outcome.text.strip()!r}"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)