From 6887ca98d790dc6c6348a75bcbd4b1ed00108ab3 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 3 Jun 2026 00:28:35 +0000 Subject: [PATCH 1/3] test(passthrough): regression test for Bedrock Content-Length override The non-streaming passthrough finalizer merges dict(fastapi_response.headers) into the outgoing response, and an empty FastAPI Response carries a default content-length: 0. Before the get_response_headers sanitization fix, that leaked through and overrode the real upstream Content-Length, so uvicorn/h11 aborted any non-empty Bedrock Invoke body and the client saw a 200 with an empty/truncated payload. Adds a unit test pinning the custom_headers exclusion path and an integration test that drives a non-empty passthrough body through a real ASGI server, which is the only level at which the Content-Length mismatch surfaces. Both fail against the pre-fix behavior and pass with the sanitization in place. --- .../test_pass_through_endpoints.py | 118 +++++++++++++++++- 1 file changed, 117 insertions(+), 1 deletion(-) diff --git a/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py b/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py index 61299e2662a..09dd0866b8f 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py @@ -1,13 +1,19 @@ +import asyncio +import contextlib import json import os +import socket import sys +import threading +import time from io import BytesIO from types import SimpleNamespace from unittest.mock import AsyncMock, MagicMock, patch import httpx import pytest -from fastapi import Request, UploadFile +import uvicorn +from fastapi import FastAPI, Request, Response, UploadFile from starlette.datastructures import Headers, QueryParams from starlette.datastructures import UploadFile as StarletteUploadFile @@ -3075,3 +3081,113 @@ def test_get_response_headers_strips_server_and_date(): assert lowered["content-type"] == "application/json" assert lowered["x-request-id"] == "req_abc" assert lowered["anthropic-ratelimit-requests-remaining"] == "100" + + +def test_get_response_headers_sanitizes_custom_headers(): + """Regression: the non-streaming passthrough finalizer passes + dict(fastapi_response.headers) as custom_headers, and an empty FastAPI + Response carries a default content-length: 0. Custom headers must run through + the same exclusion list as upstream headers so framework defaults can't + override the real body length, while genuine litellm metadata still passes.""" + upstream_headers = httpx.Headers( + {"content-type": "application/json", "content-length": "42"} + ) + custom_headers = { + "content-length": "0", + "transfer-encoding": "chunked", + "x-litellm-call-id": "call-123", + "x-litellm-version": "1.2.3", + } + + result = HttpPassThroughEndpointHelpers.get_response_headers( + upstream_headers, custom_headers=custom_headers + ) + + lowered = {k.lower(): v for k, v in result.items()} + assert "content-length" not in lowered + assert "transfer-encoding" not in lowered + assert lowered["x-litellm-call-id"] == "call-123" + assert lowered["x-litellm-version"] == "1.2.3" + + +@contextlib.contextmanager +def _serve_app(app: FastAPI): + sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + sock.bind(("127.0.0.1", 0)) + host, port = sock.getsockname() + server = uvicorn.Server(uvicorn.Config(app, log_level="warning")) + + def _run() -> None: + loop = asyncio.new_event_loop() + asyncio.set_event_loop(loop) + loop.run_until_complete(server.serve(sockets=[sock])) + + thread = threading.Thread(target=_run, daemon=True) + thread.start() + try: + deadline = time.time() + 30 + while not server.started: + if not thread.is_alive(): + raise RuntimeError("passthrough test server failed to start") + if time.time() > deadline: + raise TimeoutError("passthrough test server did not start in time") + time.sleep(0.05) + yield f"http://{host}:{port}" + finally: + server.should_exit = True + thread.join(timeout=10) + sock.close() + + +def test_bedrock_passthrough_nonempty_body_survives_real_http_serialization(): + """Regression for the v1.85.0 Bedrock passthrough Content-Length bug. + + The non-streaming finalizer merges dict(fastapi_response.headers) into the + response, and the placeholder Response carries content-length: 0. If that + leaks past get_response_headers it overrides the real upstream length, and + uvicorn/h11 aborts any non-empty body with 'Response content longer than + Content-Length', so the client sees a 200 with an empty/truncated body. Unit + assertions on the response object miss this because it only surfaces once the + ASGI server serializes over a socket, so this drives a real server.""" + upstream_body = json.dumps( + { + "output": { + "message": { + "role": "assistant", + "content": [{"text": "hello from bedrock"}], + } + } + } + ).encode() + + placeholder = Response() + placeholder.headers["x-litellm-call-id"] = "test-call-id" + + app = FastAPI() + + @app.get("/bedrock/passthrough") + async def _bedrock_passthrough() -> Response: + return Response( + content=upstream_body, + status_code=200, + headers=HttpPassThroughEndpointHelpers.get_response_headers( + headers=httpx.Headers( + { + "content-type": "application/json", + "content-length": str(len(upstream_body)), + "x-amzn-requestid": "bedrock-request-id", + } + ), + custom_headers=dict(placeholder.headers), + ), + ) + + with _serve_app(app) as base_url: + response = httpx.get(f"{base_url}/bedrock/passthrough", timeout=10) + + assert response.status_code == 200 + assert response.content == upstream_body + assert int(response.headers["content-length"]) == len(upstream_body) + assert response.headers["x-amzn-requestid"] == "bedrock-request-id" + assert response.headers["x-litellm-call-id"] == "test-call-id" From c04cb969961052a61192fe178a812e018920ed3e Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 5 Jun 2026 06:46:44 +0000 Subject: [PATCH 2/3] test(passthrough): drive real finalizer over socket in Bedrock regression The transport test reconstructed the header assembly inline, so it pinned get_response_headers over a socket but never exercised the finalizer that actually leaked content-length: 0 (base_passthrough_process_llm_request passing dict(fastapi_response.headers)). Drive that finalizer directly via a subclass that overrides the upstream call, with a real placeholder Response as the leak source, and serve its output over a real uvicorn/h11 server so a regression in the finalizer itself is caught, not just in the helper. --- .../test_pass_through_endpoints.py | 84 ++++++++++++++----- 1 file changed, 61 insertions(+), 23 deletions(-) diff --git a/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py b/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py index 09dd0866b8f..ed70c9da5dc 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py @@ -21,6 +21,7 @@ sys.path.insert( 0, os.path.abspath("../../..") ) # Adds the parent directory to the system path +from litellm.proxy.common_request_processing import ProxyBaseLLMRequestProcessing from litellm.proxy.pass_through_endpoints.pass_through_endpoints import ( HttpPassThroughEndpointHelpers, LITELLM_PASS_THROUGH_CUSTOM_BODY_STATE_KEY, @@ -3140,16 +3141,43 @@ def _serve_app(app: FastAPI): sock.close() +class _FakeUpstreamResponse: + """Stands in for the httpx upstream response the passthrough finalizer + reads. Only the attributes the finalizer touches are implemented.""" + + def __init__(self, body: bytes, headers: httpx.Headers, status_code: int = 200): + self._body = body + self.headers = headers + self.status_code = status_code + + async def aread(self) -> bytes: + return self._body + + +class _StubPassthroughProcessing(ProxyBaseLLMRequestProcessing): + """Drives the real ``base_passthrough_process_llm_request`` finalizer while + short-circuiting the upstream call via override (no monkeypatching).""" + + def __init__(self, upstream: _FakeUpstreamResponse): + super().__init__(data={}) + self._upstream = upstream + + async def base_process_llm_request(self, *args, **kwargs): # type: ignore[override] + return self._upstream + + def test_bedrock_passthrough_nonempty_body_survives_real_http_serialization(): """Regression for the v1.85.0 Bedrock passthrough Content-Length bug. - The non-streaming finalizer merges dict(fastapi_response.headers) into the - response, and the placeholder Response carries content-length: 0. If that - leaks past get_response_headers it overrides the real upstream length, and - uvicorn/h11 aborts any non-empty body with 'Response content longer than - Content-Length', so the client sees a 200 with an empty/truncated body. Unit - assertions on the response object miss this because it only surfaces once the - ASGI server serializes over a socket, so this drives a real server.""" + Drives the real ``base_passthrough_process_llm_request`` finalizer, which + merges ``dict(fastapi_response.headers)`` into the response. The placeholder + FastAPI Response carries a default content-length: 0; if that leaks past + ``get_response_headers`` it overrides the real upstream length and uvicorn/h11 + aborts any non-empty body, so the client sees a 200 with an empty/truncated + body. The symptom only surfaces once the ASGI server serializes over a + socket, so the finalizer's output is served through a real server. Asserting + on the finalizer output in-process would miss it, which is how the bug + shipped; reproducing it requires both the real finalizer and a real socket.""" upstream_body = json.dumps( { "output": { @@ -3161,27 +3189,37 @@ def test_bedrock_passthrough_nonempty_body_survives_real_http_serialization(): } ).encode() - placeholder = Response() - placeholder.headers["x-litellm-call-id"] = "test-call-id" + upstream = _FakeUpstreamResponse( + body=upstream_body, + headers=httpx.Headers( + { + "content-type": "application/json", + "content-length": str(len(upstream_body)), + "x-amzn-requestid": "bedrock-request-id", + } + ), + ) + + fastapi_response = Response() + fastapi_response.headers["x-litellm-call-id"] = "test-call-id" + + final_response = asyncio.run( + _StubPassthroughProcessing(upstream).base_passthrough_process_llm_request( + request=MagicMock(), + fastapi_response=fastapi_response, + user_api_key_dict=MagicMock(), + proxy_logging_obj=MagicMock(), + general_settings={}, + proxy_config=MagicMock(), + select_data_generator=MagicMock(), + ) + ) app = FastAPI() @app.get("/bedrock/passthrough") async def _bedrock_passthrough() -> Response: - return Response( - content=upstream_body, - status_code=200, - headers=HttpPassThroughEndpointHelpers.get_response_headers( - headers=httpx.Headers( - { - "content-type": "application/json", - "content-length": str(len(upstream_body)), - "x-amzn-requestid": "bedrock-request-id", - } - ), - custom_headers=dict(placeholder.headers), - ), - ) + return final_response with _serve_app(app) as base_url: response = httpx.get(f"{base_url}/bedrock/passthrough", timeout=10) From fe574de7ab768c32835f6856771de3c1dc58a7c5 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 8 Jun 2026 18:59:32 +0000 Subject: [PATCH 3/3] test(passthrough): serialize Bedrock Content-Length regression in-process tests/test_litellm is documented as mock-only (tests/test_litellm/readme.md), but the Bedrock passthrough regression test bound a real loopback socket and ran a live uvicorn server in a background thread. Drive the same finalizer output through an in-process ASGI server (httpx.ASGITransport) instead, asserting the recomputed Content-Length the client receives against the served body. The test still fails against the pre-#29120 leak, where the placeholder Response's content-length: 0 overrides the real upstream length, and passes once get_response_headers sanitizes custom headers, without opening a real socket. --- .../test_pass_through_endpoints.py | 74 ++++++------------- 1 file changed, 23 insertions(+), 51 deletions(-) diff --git a/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py b/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py index ed70c9da5dc..a8696b11c26 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py @@ -1,18 +1,13 @@ import asyncio -import contextlib import json import os -import socket import sys -import threading -import time from io import BytesIO from types import SimpleNamespace from unittest.mock import AsyncMock, MagicMock, patch import httpx import pytest -import uvicorn from fastapi import FastAPI, Request, Response, UploadFile from starlette.datastructures import Headers, QueryParams from starlette.datastructures import UploadFile as StarletteUploadFile @@ -3111,36 +3106,6 @@ def test_get_response_headers_sanitizes_custom_headers(): assert lowered["x-litellm-version"] == "1.2.3" -@contextlib.contextmanager -def _serve_app(app: FastAPI): - sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM) - sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) - sock.bind(("127.0.0.1", 0)) - host, port = sock.getsockname() - server = uvicorn.Server(uvicorn.Config(app, log_level="warning")) - - def _run() -> None: - loop = asyncio.new_event_loop() - asyncio.set_event_loop(loop) - loop.run_until_complete(server.serve(sockets=[sock])) - - thread = threading.Thread(target=_run, daemon=True) - thread.start() - try: - deadline = time.time() + 30 - while not server.started: - if not thread.is_alive(): - raise RuntimeError("passthrough test server failed to start") - if time.time() > deadline: - raise TimeoutError("passthrough test server did not start in time") - time.sleep(0.05) - yield f"http://{host}:{port}" - finally: - server.should_exit = True - thread.join(timeout=10) - sock.close() - - class _FakeUpstreamResponse: """Stands in for the httpx upstream response the passthrough finalizer reads. Only the attributes the finalizer touches are implemented.""" @@ -3166,18 +3131,19 @@ class _StubPassthroughProcessing(ProxyBaseLLMRequestProcessing): return self._upstream -def test_bedrock_passthrough_nonempty_body_survives_real_http_serialization(): +def test_bedrock_passthrough_nonempty_body_survives_http_serialization(): """Regression for the v1.85.0 Bedrock passthrough Content-Length bug. Drives the real ``base_passthrough_process_llm_request`` finalizer, which merges ``dict(fastapi_response.headers)`` into the response. The placeholder FastAPI Response carries a default content-length: 0; if that leaks past - ``get_response_headers`` it overrides the real upstream length and uvicorn/h11 - aborts any non-empty body, so the client sees a 200 with an empty/truncated - body. The symptom only surfaces once the ASGI server serializes over a - socket, so the finalizer's output is served through a real server. Asserting - on the finalizer output in-process would miss it, which is how the bug - shipped; reproducing it requires both the real finalizer and a real socket.""" + ``get_response_headers`` it overrides the real upstream length, so a + non-empty Bedrock body is served with a Content-Length that no longer matches + it. The finalizer's output is serialized through an in-process ASGI server + (httpx.ASGITransport) and the Content-Length the client receives is asserted + against the served body, the mismatch the object-level assertion in #27412 + didn't check. This folder is mock-only (see tests/test_litellm/readme.md), so + the test stays in-process instead of binding a real socket.""" upstream_body = json.dumps( { "output": { @@ -3203,8 +3169,10 @@ def test_bedrock_passthrough_nonempty_body_survives_real_http_serialization(): fastapi_response = Response() fastapi_response.headers["x-litellm-call-id"] = "test-call-id" - final_response = asyncio.run( - _StubPassthroughProcessing(upstream).base_passthrough_process_llm_request( + async def _drive_and_serve() -> httpx.Response: + final_response = await _StubPassthroughProcessing( + upstream + ).base_passthrough_process_llm_request( request=MagicMock(), fastapi_response=fastapi_response, user_api_key_dict=MagicMock(), @@ -3213,16 +3181,20 @@ def test_bedrock_passthrough_nonempty_body_survives_real_http_serialization(): proxy_config=MagicMock(), select_data_generator=MagicMock(), ) - ) - app = FastAPI() + app = FastAPI() - @app.get("/bedrock/passthrough") - async def _bedrock_passthrough() -> Response: - return final_response + @app.get("/bedrock/passthrough") + async def _bedrock_passthrough() -> Response: + return final_response - with _serve_app(app) as base_url: - response = httpx.get(f"{base_url}/bedrock/passthrough", timeout=10) + transport = httpx.ASGITransport(app=app) + async with httpx.AsyncClient( + transport=transport, base_url="http://bedrock-passthrough" + ) as client: + return await client.get("/bedrock/passthrough") + + response = asyncio.run(_drive_and_serve()) assert response.status_code == 200 assert response.content == upstream_body