fix: prevent long scans from hanging or dying on fd exhaustion

Two independent failure modes surfaced on long multi-agent scans:

1. Too many open files (OSError 24). A long scan accumulates file
   descriptors (httpx pools, docker socket polling, per-subagent
   resources). The process never raised its soft RLIMIT_NOFILE, so on
   macOS (default soft limit 256) a long scan blew past the ceiling and
   crashed at whatever fd-opening call lost the race. Raise the soft
   limit toward the hard limit at startup.

2. Agents stuck at "Starting agent...". LLM_TIMEOUT was only applied to
   the startup warm-up call, never to agent-loop calls. Models on the
   openai/ prefix go through the SDK's OpenAI provider whose default
   timeout is 600s; as an httpx read timeout that parks an agent for ten
   minutes on a stalled stream. Install a default AsyncOpenAI client
   whose read timeout is bounded by LLM_TIMEOUT so a stalled stream
   raises and the existing retry policy recovers the agent.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Felix 2026-06-29 17:59:25 +08:00
parent 7141ccff62
commit 2833bd34e5
2 changed files with 59 additions and 1 deletions

View file

@ -3,7 +3,7 @@
from __future__ import annotations
import os
from typing import TYPE_CHECKING
from typing import TYPE_CHECKING, Any
from agents import set_default_openai_api, set_default_openai_key, set_tracing_disabled
from agents.models.multi_provider import MultiProvider
@ -75,6 +75,37 @@ def configure_sdk_model_defaults(settings: Settings) -> None:
set_default_openai_api("chat_completions")
else:
set_default_openai_api("responses")
_configure_openai_client_timeout(llm)
def _configure_openai_client_timeout(llm: Any) -> None:
"""Install a default AsyncOpenAI client whose read timeout is bounded.
Models on the ``openai/`` prefix (e.g. ``openai/gpt-5.5`` against a
custom ``api_base``) go through the SDK's OpenAI provider, NOT
litellm — so ``litellm.request_timeout`` does not apply to them. The
OpenAI SDK's default timeout is 600s, and as a *read* timeout that
means a stalled stream (server stops sending bytes without closing
the socket) parks the agent for 10 minutes — surfaced in the TUI as
an agent stuck at "Starting agent...".
Passing a float ``timeout`` makes the underlying httpx read timeout an
*idle* timeout: it resets on every received chunk, so a healthy long
generation is never cut off, but a truly stalled stream raises after
``LLM_TIMEOUT`` seconds and the SDK retry policy recovers the agent.
The client carries the key/base_url because ``set_default_openai_client``
takes precedence over ``set_default_openai_key``.
"""
if not llm.api_key or llm.timeout <= 0:
return
from agents import set_default_openai_client
from openai import AsyncOpenAI
client_kwargs: dict[str, Any] = {"api_key": llm.api_key, "timeout": float(llm.timeout)}
if llm.api_base:
client_kwargs["base_url"] = llm.api_base
set_default_openai_client(AsyncOpenAI(**client_kwargs), use_for_tracing=False)
def _mirror_api_key_to_provider_env(model_name: str | None, api_key: str) -> None:

View file

@ -737,8 +737,35 @@ def pull_docker_image() -> None:
console.print()
def _raise_open_files_limit() -> None:
"""Raise the soft RLIMIT_NOFILE toward the hard limit.
A long multi-agent scan accumulates file descriptors (litellm httpx
pools, docker socket polling, per-subagent resources). On macOS the
default soft limit is just 256, which a long scan blows past and then
dies with ``OSError: [Errno 24] Too many open files`` — surfacing at
whatever random fd-opening call happens to lose the race. Bumping the
soft limit to the hard limit removes that ceiling. Not available on
Windows, where ``resource`` is absent.
"""
try:
import resource
except ImportError:
return
soft, hard = resource.getrlimit(resource.RLIMIT_NOFILE)
target = hard if hard != resource.RLIM_INFINITY else 1_048_576
if soft >= target:
return
try:
resource.setrlimit(resource.RLIMIT_NOFILE, (target, hard))
except (ValueError, OSError):
logger.debug("Could not raise RLIMIT_NOFILE from %s to %s", soft, target, exc_info=True)
def main() -> None:
configure_dependency_logging()
_raise_open_files_limit()
if sys.platform == "win32":
asyncio.set_event_loop_policy(asyncio.WindowsSelectorEventLoopPolicy())