mirror of
https://github.com/open-webui/open-webui.git
synced 2026-10-03 02:23:51 +00:00
feat: configurable aiohttp read_bufsize via AIOHTTP_READ_BUFSIZE env
The shared aiohttp ClientSession was created with the library default read_bufsize (64 KiB). Some upstream providers emit a single SSE chunk larger than that ceiling, which fails the stream with HTTP 400 and the message "Got more than 131072 bytes when reading: b'data: ...'". Known triggers seen in the wild: - OpenRouter routing reasoning models (Kimi K2.6, R1-style) through providers that batch the full reasoning trace into one chunk (Novita, WandB, SiliconFlow) - Image-generation responses returned as a single large SSE message (related: discussion #24543) Fix: expose AIOHTTP_READ_BUFSIZE (default 1 MiB) and pass it through to ClientSession. Users hitting unusually large chunks can raise it further without code changes. No behavior change at the default; the 1 MiB default is conservative enough to handle the cases above while still rejecting truly runaway payloads.
This commit is contained in:
parent
1527eb6e01
commit
d1a8722545
2 changed files with 23 additions and 1 deletions
|
|
@ -578,6 +578,20 @@ try:
|
|||
except ValueError:
|
||||
AIOHTTP_POOL_DNS_TTL = 300
|
||||
|
||||
# Max size of a single chunk read from an SSE/HTTP stream, in bytes.
|
||||
# aiohttp's default is 64 KiB which can be exceeded when an upstream provider
|
||||
# emits a single very large SSE chunk (e.g. OpenRouter sometimes batches a
|
||||
# reasoning trace or tool-call payload into one chunk, or an image-generation
|
||||
# response is returned in one message). Raising this avoids
|
||||
# `Got more than N bytes when reading` 400 errors. Default 1 MiB.
|
||||
AIOHTTP_READ_BUFSIZE = os.getenv('AIOHTTP_READ_BUFSIZE', str(1024 * 1024))
|
||||
try:
|
||||
AIOHTTP_READ_BUFSIZE = int(AIOHTTP_READ_BUFSIZE)
|
||||
if AIOHTTP_READ_BUFSIZE <= 0:
|
||||
AIOHTTP_READ_BUFSIZE = 1024 * 1024
|
||||
except ValueError:
|
||||
AIOHTTP_READ_BUFSIZE = 1024 * 1024
|
||||
|
||||
RAG_EMBEDDING_TIMEOUT = os.getenv('RAG_EMBEDDING_TIMEOUT', '')
|
||||
|
||||
if RAG_EMBEDDING_TIMEOUT == '':
|
||||
|
|
|
|||
|
|
@ -9,6 +9,11 @@ All pool parameters are configurable via environment variables:
|
|||
- AIOHTTP_POOL_CONNECTIONS (default 100) — max total connections
|
||||
- AIOHTTP_POOL_CONNECTIONS_PER_HOST (default 30) — per-host limit
|
||||
- AIOHTTP_POOL_DNS_TTL (default 300) — DNS cache TTL in seconds
|
||||
- AIOHTTP_READ_BUFSIZE (default 1048576) — per-chunk read buffer size in
|
||||
bytes. Raise this when an upstream emits very large SSE chunks (e.g.
|
||||
reasoning-trace batching by some OpenRouter providers, or single-message
|
||||
image-generation responses) which otherwise fail with
|
||||
"Got more than N bytes when reading".
|
||||
|
||||
Usage:
|
||||
from open_webui.utils.session_pool import get_session, cleanup_response
|
||||
|
|
@ -32,6 +37,7 @@ from open_webui.env import (
|
|||
AIOHTTP_POOL_CONNECTIONS,
|
||||
AIOHTTP_POOL_CONNECTIONS_PER_HOST,
|
||||
AIOHTTP_POOL_DNS_TTL,
|
||||
AIOHTTP_READ_BUFSIZE,
|
||||
)
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
|
@ -61,12 +67,14 @@ async def get_session() -> aiohttp.ClientSession:
|
|||
connector=connector,
|
||||
timeout=timeout,
|
||||
trust_env=True,
|
||||
read_bufsize=AIOHTTP_READ_BUFSIZE,
|
||||
)
|
||||
log.info(
|
||||
'Created shared aiohttp session pool (limit=%s, per_host=%s, dns_ttl=%d)',
|
||||
'Created shared aiohttp session pool (limit=%s, per_host=%s, dns_ttl=%d, read_bufsize=%d)',
|
||||
AIOHTTP_POOL_CONNECTIONS or 'unlimited',
|
||||
AIOHTTP_POOL_CONNECTIONS_PER_HOST or 'unlimited',
|
||||
AIOHTTP_POOL_DNS_TTL,
|
||||
AIOHTTP_READ_BUFSIZE,
|
||||
)
|
||||
return _session
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue