mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
fix(harness): preserve PDF trailers in benchmark fixtures
This commit is contained in:
parent
e73d21508d
commit
af1c8eb5ed
3 changed files with 10 additions and 5 deletions
|
|
@ -45,7 +45,7 @@ The seed cassette stays under `e2e_parity/sdk/ocr/fixtures/data`. The benchmark
|
|||
| response_medium | 32 KiB | 16 |
|
||||
| response_large | 32 KiB | 128 |
|
||||
|
||||
Request variants add PDF comment padding before the EOF marker, preserving existing object offsets. The SDK sends base64 plus JSON framing, so wire request sizes exceed the document sizes above. Response variants repeat recorded pages with contiguous indexes and adjusted usage. They exercise realistic response structure, but their page count intentionally varies independently of the input PDF's content
|
||||
Request variants add PDF comment padding before the final `startxref` marker, preserving existing object offsets and the EOF trailer. The SDK sends base64 plus JSON framing, so wire request sizes exceed the document sizes above. Response variants repeat recorded pages with contiguous indexes and adjusted usage. They exercise realistic response structure, but their page count intentionally varies independently of the input PDF's content
|
||||
|
||||
## Measurements
|
||||
|
||||
|
|
|
|||
|
|
@ -64,8 +64,8 @@ def test_pdf_padding_preserves_existing_offsets_and_exact_size() -> None:
|
|||
seed: Final = b"%PDF-1.7\n1 0 obj\n<<>>\nendobj\nstartxref\n9\n%%EOF\n"
|
||||
padded: Final = padded_pdf(seed, 1024)
|
||||
assert len(padded) == 1024
|
||||
assert padded.startswith(seed.split(b"%%EOF")[0])
|
||||
assert padded.endswith(b"\n%%EOF\n")
|
||||
assert padded.startswith(seed.split(b"startxref")[0])
|
||||
assert padded.endswith(b"\nstartxref\n9\n%%EOF\n")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("arguments", (("--iterations=0",), ("--warmup=0",), ("--route=chat",), ("--profile=unknown",)))
|
||||
|
|
|
|||
|
|
@ -47,8 +47,13 @@ def profile_sizes(profile: Profile) -> tuple[int, int]:
|
|||
|
||||
|
||||
def padded_pdf(document: bytes, size: int) -> bytes:
|
||||
prefix, marker, suffix = document.rpartition(b"%%EOF")
|
||||
if not marker or not document.startswith(b"%PDF-") or size < len(document) + 3:
|
||||
prefix, marker, suffix = document.rpartition(b"startxref")
|
||||
if (
|
||||
not marker
|
||||
or not suffix.rstrip().endswith(b"%%EOF")
|
||||
or not document.startswith(b"%PDF-")
|
||||
or size < len(document) + 3
|
||||
):
|
||||
raise ValueError("expected a PDF seed smaller than the requested document size")
|
||||
return prefix + b"%" + b"x" * (size - len(document) - 2) + b"\n" + marker + suffix
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue