From cdad5dd0a8e488d91311bf301a36afa5004306be Mon Sep 17 00:00:00 2001 From: VoidChecksum Date: Tue, 9 Jun 2026 00:36:08 +0000 Subject: [PATCH] fix(llm): honor LLM_TIMEOUT during scans, not just warm-up LLM_TIMEOUT is documented (docs/advanced/configuration.mdx) as the request timeout for LLM calls and is offered as the workaround for slow local models, but it was only applied to the warm-up call in strix/interface/main.py. Scan LLM calls go through the SDK's LiteLLM model, which invokes litellm.acompletion without a timeout and falls back to LiteLLM's module default (6000s) -- so `export LLM_TIMEOUT=600` had no effect on the actual run. Set litellm.request_timeout from settings in configure_sdk_model_defaults so the documented setting takes effect for LiteLLM-routed scans. Fixes #426 --- strix/config/models.py | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/strix/config/models.py b/strix/config/models.py index f5826433..8b36e5cd 100644 --- a/strix/config/models.py +++ b/strix/config/models.py @@ -63,6 +63,7 @@ def configure_sdk_model_defaults(settings: Settings) -> None: llm = settings.llm set_tracing_disabled(True) _configure_litellm_compatibility() + _configure_litellm_request_timeout(llm.timeout) if llm.api_key: set_default_openai_key(llm.api_key, use_for_tracing=False) _configure_litellm_default("api_key", llm.api_key) @@ -128,6 +129,23 @@ def _configure_litellm_default(name: str, value: str) -> None: setattr(litellm, name, value) +def _configure_litellm_request_timeout(timeout: int) -> None: + """Apply the configured ``LLM_TIMEOUT`` to LiteLLM-routed scan calls. + + The SDK's LiteLLM model invokes ``litellm.acompletion`` without an explicit + per-request timeout, so without this it falls back to LiteLLM's module + default and the documented ``LLM_TIMEOUT`` only affected the warm-up call + (``strix/interface/main.py``). Setting the module-level default makes + ``export LLM_TIMEOUT=600`` take effect for the actual scan, restoring the + documented behavior for slow local / self-hosted models. + """ + import litellm + + # litellm doesn't re-export request_timeout in its public surface, but it is + # the module-level default it reads for each acompletion call. + litellm.request_timeout = timeout # type: ignore[attr-defined] + + def uses_chat_completions_tool_schema(model_name: str, settings: Settings) -> bool: """Return whether the resolved SDK route can only receive JSON function tools.""" model = model_name.strip().lower()