mirror of
https://github.com/usestrix/strix.git
synced 2026-10-11 03:37:54 +00:00
fix: prevent long scans from hanging or dying on fd exhaustion
Two independent failure modes surfaced on long multi-agent scans: 1. Too many open files (OSError 24). A long scan accumulates file descriptors (httpx pools, docker socket polling, per-subagent resources). The process never raised its soft RLIMIT_NOFILE, so on macOS (default soft limit 256) a long scan blew past the ceiling and crashed at whatever fd-opening call lost the race. Raise the soft limit toward the hard limit at startup. 2. Agents stuck at "Starting agent...". LLM_TIMEOUT was only applied to the startup warm-up call, never to agent-loop calls. Models on the openai/ prefix go through the SDK's OpenAI provider whose default timeout is 600s; as an httpx read timeout that parks an agent for ten minutes on a stalled stream. Install a default AsyncOpenAI client whose read timeout is bounded by LLM_TIMEOUT so a stalled stream raises and the existing retry policy recovers the agent. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
7141ccff62
commit
2833bd34e5
2 changed files with 59 additions and 1 deletions
|
|
@ -3,7 +3,7 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from typing import TYPE_CHECKING
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from agents import set_default_openai_api, set_default_openai_key, set_tracing_disabled
|
||||
from agents.models.multi_provider import MultiProvider
|
||||
|
|
@ -75,6 +75,37 @@ def configure_sdk_model_defaults(settings: Settings) -> None:
|
|||
set_default_openai_api("chat_completions")
|
||||
else:
|
||||
set_default_openai_api("responses")
|
||||
_configure_openai_client_timeout(llm)
|
||||
|
||||
|
||||
def _configure_openai_client_timeout(llm: Any) -> None:
|
||||
"""Install a default AsyncOpenAI client whose read timeout is bounded.
|
||||
|
||||
Models on the ``openai/`` prefix (e.g. ``openai/gpt-5.5`` against a
|
||||
custom ``api_base``) go through the SDK's OpenAI provider, NOT
|
||||
litellm — so ``litellm.request_timeout`` does not apply to them. The
|
||||
OpenAI SDK's default timeout is 600s, and as a *read* timeout that
|
||||
means a stalled stream (server stops sending bytes without closing
|
||||
the socket) parks the agent for 10 minutes — surfaced in the TUI as
|
||||
an agent stuck at "Starting agent...".
|
||||
|
||||
Passing a float ``timeout`` makes the underlying httpx read timeout an
|
||||
*idle* timeout: it resets on every received chunk, so a healthy long
|
||||
generation is never cut off, but a truly stalled stream raises after
|
||||
``LLM_TIMEOUT`` seconds and the SDK retry policy recovers the agent.
|
||||
|
||||
The client carries the key/base_url because ``set_default_openai_client``
|
||||
takes precedence over ``set_default_openai_key``.
|
||||
"""
|
||||
if not llm.api_key or llm.timeout <= 0:
|
||||
return
|
||||
from agents import set_default_openai_client
|
||||
from openai import AsyncOpenAI
|
||||
|
||||
client_kwargs: dict[str, Any] = {"api_key": llm.api_key, "timeout": float(llm.timeout)}
|
||||
if llm.api_base:
|
||||
client_kwargs["base_url"] = llm.api_base
|
||||
set_default_openai_client(AsyncOpenAI(**client_kwargs), use_for_tracing=False)
|
||||
|
||||
|
||||
def _mirror_api_key_to_provider_env(model_name: str | None, api_key: str) -> None:
|
||||
|
|
|
|||
|
|
@ -737,8 +737,35 @@ def pull_docker_image() -> None:
|
|||
console.print()
|
||||
|
||||
|
||||
def _raise_open_files_limit() -> None:
|
||||
"""Raise the soft RLIMIT_NOFILE toward the hard limit.
|
||||
|
||||
A long multi-agent scan accumulates file descriptors (litellm httpx
|
||||
pools, docker socket polling, per-subagent resources). On macOS the
|
||||
default soft limit is just 256, which a long scan blows past and then
|
||||
dies with ``OSError: [Errno 24] Too many open files`` — surfacing at
|
||||
whatever random fd-opening call happens to lose the race. Bumping the
|
||||
soft limit to the hard limit removes that ceiling. Not available on
|
||||
Windows, where ``resource`` is absent.
|
||||
"""
|
||||
try:
|
||||
import resource
|
||||
except ImportError:
|
||||
return
|
||||
|
||||
soft, hard = resource.getrlimit(resource.RLIMIT_NOFILE)
|
||||
target = hard if hard != resource.RLIM_INFINITY else 1_048_576
|
||||
if soft >= target:
|
||||
return
|
||||
try:
|
||||
resource.setrlimit(resource.RLIMIT_NOFILE, (target, hard))
|
||||
except (ValueError, OSError):
|
||||
logger.debug("Could not raise RLIMIT_NOFILE from %s to %s", soft, target, exc_info=True)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
configure_dependency_logging()
|
||||
_raise_open_files_limit()
|
||||
|
||||
if sys.platform == "win32":
|
||||
asyncio.set_event_loop_policy(asyncio.WindowsSelectorEventLoopPolicy())
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue