diff --git a/strix/config/models.py b/strix/config/models.py index f5826433..8b36e5cd 100644 --- a/strix/config/models.py +++ b/strix/config/models.py @@ -63,6 +63,7 @@ def configure_sdk_model_defaults(settings: Settings) -> None: llm = settings.llm set_tracing_disabled(True) _configure_litellm_compatibility() + _configure_litellm_request_timeout(llm.timeout) if llm.api_key: set_default_openai_key(llm.api_key, use_for_tracing=False) _configure_litellm_default("api_key", llm.api_key) @@ -128,6 +129,23 @@ def _configure_litellm_default(name: str, value: str) -> None: setattr(litellm, name, value) +def _configure_litellm_request_timeout(timeout: int) -> None: + """Apply the configured ``LLM_TIMEOUT`` to LiteLLM-routed scan calls. + + The SDK's LiteLLM model invokes ``litellm.acompletion`` without an explicit + per-request timeout, so without this it falls back to LiteLLM's module + default and the documented ``LLM_TIMEOUT`` only affected the warm-up call + (``strix/interface/main.py``). Setting the module-level default makes + ``export LLM_TIMEOUT=600`` take effect for the actual scan, restoring the + documented behavior for slow local / self-hosted models. + """ + import litellm + + # litellm doesn't re-export request_timeout in its public surface, but it is + # the module-level default it reads for each acompletion call. + litellm.request_timeout = timeout # type: ignore[attr-defined] + + def uses_chat_completions_tool_schema(model_name: str, settings: Settings) -> bool: """Return whether the resolved SDK route can only receive JSON function tools.""" model = model_name.strip().lower()