fix(harness): preserve PDF trailers in benchmark fixtures

This commit is contained in:
Yujong Lee 2026-09-05 11:55:29 -07:00
parent e73d21508d
commit af1c8eb5ed
3 changed files with 10 additions and 5 deletions

View file

@ -45,7 +45,7 @@ The seed cassette stays under `e2e_parity/sdk/ocr/fixtures/data`. The benchmark
| response_medium | 32 KiB | 16 |
| response_large | 32 KiB | 128 |
Request variants add PDF comment padding before the EOF marker, preserving existing object offsets. The SDK sends base64 plus JSON framing, so wire request sizes exceed the document sizes above. Response variants repeat recorded pages with contiguous indexes and adjusted usage. They exercise realistic response structure, but their page count intentionally varies independently of the input PDF's content
Request variants add PDF comment padding before the final `startxref` marker, preserving existing object offsets and the EOF trailer. The SDK sends base64 plus JSON framing, so wire request sizes exceed the document sizes above. Response variants repeat recorded pages with contiguous indexes and adjusted usage. They exercise realistic response structure, but their page count intentionally varies independently of the input PDF's content
## Measurements

View file

@ -64,8 +64,8 @@ def test_pdf_padding_preserves_existing_offsets_and_exact_size() -> None:
seed: Final = b"%PDF-1.7\n1 0 obj\n<<>>\nendobj\nstartxref\n9\n%%EOF\n"
padded: Final = padded_pdf(seed, 1024)
assert len(padded) == 1024
assert padded.startswith(seed.split(b"%%EOF")[0])
assert padded.endswith(b"\n%%EOF\n")
assert padded.startswith(seed.split(b"startxref")[0])
assert padded.endswith(b"\nstartxref\n9\n%%EOF\n")
@pytest.mark.parametrize("arguments", (("--iterations=0",), ("--warmup=0",), ("--route=chat",), ("--profile=unknown",)))

View file

@ -47,8 +47,13 @@ def profile_sizes(profile: Profile) -> tuple[int, int]:
def padded_pdf(document: bytes, size: int) -> bytes:
prefix, marker, suffix = document.rpartition(b"%%EOF")
if not marker or not document.startswith(b"%PDF-") or size < len(document) + 3:
prefix, marker, suffix = document.rpartition(b"startxref")
if (
not marker
or not suffix.rstrip().endswith(b"%%EOF")
or not document.startswith(b"%PDF-")
or size < len(document) + 3
):
raise ValueError("expected a PDF seed smaller than the requested document size")
return prefix + b"%" + b"x" * (size - len(document) - 2) + b"\n" + marker + suffix