mirror of
https://github.com/usestrix/strix.git
synced 2026-10-05 02:41:38 +00:00
fix(llm): honor LLM_TIMEOUT during scans, not just warm-up
LLM_TIMEOUT is documented (docs/advanced/configuration.mdx) as the request timeout for LLM calls and is offered as the workaround for slow local models, but it was only applied to the warm-up call in strix/interface/main.py. Scan LLM calls go through the SDK's LiteLLM model, which invokes litellm.acompletion without a timeout and falls back to LiteLLM's module default (6000s) -- so `export LLM_TIMEOUT=600` had no effect on the actual run. Set litellm.request_timeout from settings in configure_sdk_model_defaults so the documented setting takes effect for LiteLLM-routed scans. Fixes #426
This commit is contained in:
parent
760803f983
commit
cdad5dd0a8
1 changed files with 18 additions and 0 deletions
|
|
@ -63,6 +63,7 @@ def configure_sdk_model_defaults(settings: Settings) -> None:
|
|||
llm = settings.llm
|
||||
set_tracing_disabled(True)
|
||||
_configure_litellm_compatibility()
|
||||
_configure_litellm_request_timeout(llm.timeout)
|
||||
if llm.api_key:
|
||||
set_default_openai_key(llm.api_key, use_for_tracing=False)
|
||||
_configure_litellm_default("api_key", llm.api_key)
|
||||
|
|
@ -128,6 +129,23 @@ def _configure_litellm_default(name: str, value: str) -> None:
|
|||
setattr(litellm, name, value)
|
||||
|
||||
|
||||
def _configure_litellm_request_timeout(timeout: int) -> None:
|
||||
"""Apply the configured ``LLM_TIMEOUT`` to LiteLLM-routed scan calls.
|
||||
|
||||
The SDK's LiteLLM model invokes ``litellm.acompletion`` without an explicit
|
||||
per-request timeout, so without this it falls back to LiteLLM's module
|
||||
default and the documented ``LLM_TIMEOUT`` only affected the warm-up call
|
||||
(``strix/interface/main.py``). Setting the module-level default makes
|
||||
``export LLM_TIMEOUT=600`` take effect for the actual scan, restoring the
|
||||
documented behavior for slow local / self-hosted models.
|
||||
"""
|
||||
import litellm
|
||||
|
||||
# litellm doesn't re-export request_timeout in its public surface, but it is
|
||||
# the module-level default it reads for each acompletion call.
|
||||
litellm.request_timeout = timeout # type: ignore[attr-defined]
|
||||
|
||||
|
||||
def uses_chat_completions_tool_schema(model_name: str, settings: Settings) -> bool:
|
||||
"""Return whether the resolved SDK route can only receive JSON function tools."""
|
||||
model = model_name.strip().lower()
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue