From 2833bd34e58db803310c14fac9133a9047dd2eb5 Mon Sep 17 00:00:00 2001 From: Felix <0xfelixli@gmail.com> Date: Mon, 29 Jun 2026 17:59:25 +0800 Subject: [PATCH] fix: prevent long scans from hanging or dying on fd exhaustion Two independent failure modes surfaced on long multi-agent scans: 1. Too many open files (OSError 24). A long scan accumulates file descriptors (httpx pools, docker socket polling, per-subagent resources). The process never raised its soft RLIMIT_NOFILE, so on macOS (default soft limit 256) a long scan blew past the ceiling and crashed at whatever fd-opening call lost the race. Raise the soft limit toward the hard limit at startup. 2. Agents stuck at "Starting agent...". LLM_TIMEOUT was only applied to the startup warm-up call, never to agent-loop calls. Models on the openai/ prefix go through the SDK's OpenAI provider whose default timeout is 600s; as an httpx read timeout that parks an agent for ten minutes on a stalled stream. Install a default AsyncOpenAI client whose read timeout is bounded by LLM_TIMEOUT so a stalled stream raises and the existing retry policy recovers the agent. Co-Authored-By: Claude Opus 4.8 (1M context) --- strix/config/models.py | 33 ++++++++++++++++++++++++++++++++- strix/interface/main.py | 27 +++++++++++++++++++++++++++ 2 files changed, 59 insertions(+), 1 deletion(-) diff --git a/strix/config/models.py b/strix/config/models.py index 7f6227f3..396701ae 100644 --- a/strix/config/models.py +++ b/strix/config/models.py @@ -3,7 +3,7 @@ from __future__ import annotations import os -from typing import TYPE_CHECKING +from typing import TYPE_CHECKING, Any from agents import set_default_openai_api, set_default_openai_key, set_tracing_disabled from agents.models.multi_provider import MultiProvider @@ -75,6 +75,37 @@ def configure_sdk_model_defaults(settings: Settings) -> None: set_default_openai_api("chat_completions") else: set_default_openai_api("responses") + _configure_openai_client_timeout(llm) + + +def _configure_openai_client_timeout(llm: Any) -> None: + """Install a default AsyncOpenAI client whose read timeout is bounded. + + Models on the ``openai/`` prefix (e.g. ``openai/gpt-5.5`` against a + custom ``api_base``) go through the SDK's OpenAI provider, NOT + litellm — so ``litellm.request_timeout`` does not apply to them. The + OpenAI SDK's default timeout is 600s, and as a *read* timeout that + means a stalled stream (server stops sending bytes without closing + the socket) parks the agent for 10 minutes — surfaced in the TUI as + an agent stuck at "Starting agent...". + + Passing a float ``timeout`` makes the underlying httpx read timeout an + *idle* timeout: it resets on every received chunk, so a healthy long + generation is never cut off, but a truly stalled stream raises after + ``LLM_TIMEOUT`` seconds and the SDK retry policy recovers the agent. + + The client carries the key/base_url because ``set_default_openai_client`` + takes precedence over ``set_default_openai_key``. + """ + if not llm.api_key or llm.timeout <= 0: + return + from agents import set_default_openai_client + from openai import AsyncOpenAI + + client_kwargs: dict[str, Any] = {"api_key": llm.api_key, "timeout": float(llm.timeout)} + if llm.api_base: + client_kwargs["base_url"] = llm.api_base + set_default_openai_client(AsyncOpenAI(**client_kwargs), use_for_tracing=False) def _mirror_api_key_to_provider_env(model_name: str | None, api_key: str) -> None: diff --git a/strix/interface/main.py b/strix/interface/main.py index 4eae0527..49e0d2d9 100644 --- a/strix/interface/main.py +++ b/strix/interface/main.py @@ -737,8 +737,35 @@ def pull_docker_image() -> None: console.print() +def _raise_open_files_limit() -> None: + """Raise the soft RLIMIT_NOFILE toward the hard limit. + + A long multi-agent scan accumulates file descriptors (litellm httpx + pools, docker socket polling, per-subagent resources). On macOS the + default soft limit is just 256, which a long scan blows past and then + dies with ``OSError: [Errno 24] Too many open files`` — surfacing at + whatever random fd-opening call happens to lose the race. Bumping the + soft limit to the hard limit removes that ceiling. Not available on + Windows, where ``resource`` is absent. + """ + try: + import resource + except ImportError: + return + + soft, hard = resource.getrlimit(resource.RLIMIT_NOFILE) + target = hard if hard != resource.RLIM_INFINITY else 1_048_576 + if soft >= target: + return + try: + resource.setrlimit(resource.RLIMIT_NOFILE, (target, hard)) + except (ValueError, OSError): + logger.debug("Could not raise RLIMIT_NOFILE from %s to %s", soft, target, exc_info=True) + + def main() -> None: configure_dependency_logging() + _raise_open_files_limit() if sys.platform == "win32": asyncio.set_event_loop_policy(asyncio.WindowsSelectorEventLoopPolicy())