From eb4bff621a3733d1dbf4e1f93bc39d653a72a999 Mon Sep 17 00:00:00 2001 From: VoidChecksum Date: Tue, 9 Jun 2026 05:07:30 +0000 Subject: [PATCH] feat(llm): expose temperature, top_p, and max_tokens settings MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add optional generation parameters so users can steer model behavior — particularly useful for local / OpenAI-compatible models that need a lower temperature for steadier tool calling (see #514). - STRIX_LLM_TEMPERATURE, STRIX_LLM_TOP_P, STRIX_LLM_MAX_TOKENS (all unset by default -> provider defaults, so no behavior change). - Threaded through make_model_settings into the SDK ModelSettings used for the scan. Params a given model rejects are dropped automatically (litellm.drop_params is already enabled). - Documented in docs/advanced/configuration.mdx. Closes #514 --- docs/advanced/configuration.mdx | 12 ++++++++++++ strix/config/settings.py | 3 +++ strix/core/inputs.py | 6 ++++++ strix/core/runner.py | 3 +++ 4 files changed, 24 insertions(+) diff --git a/docs/advanced/configuration.mdx b/docs/advanced/configuration.mdx index 9ab7f017..e85e4716 100644 --- a/docs/advanced/configuration.mdx +++ b/docs/advanced/configuration.mdx @@ -31,6 +31,18 @@ Configure Strix using environment variables or a config file. Control thinking effort for reasoning models. Valid values: `none`, `minimal`, `low`, `medium`, `high`, `xhigh`. Defaults to `medium` for quick scan mode. + + Sampling temperature passed to the model. Unset by default (uses the provider default). Useful for steadier tool use on local/OpenAI-compatible models. Parameters unsupported by a given model are dropped automatically. + + + + Nucleus sampling `top_p` passed to the model. Unset by default. + + + + Maximum number of tokens to generate per LLM response. Unset by default (uses the provider default). + + Timeout in seconds for memory compression operations (context summarization). diff --git a/strix/config/settings.py b/strix/config/settings.py index 1458e1ff..578c3f44 100644 --- a/strix/config/settings.py +++ b/strix/config/settings.py @@ -37,6 +37,9 @@ class LlmSettings(BaseSettings): ) reasoning_effort: ReasoningEffort = Field(default="high", alias="STRIX_REASONING_EFFORT") timeout: int = Field(default=300, alias="LLM_TIMEOUT") + temperature: float | None = Field(default=None, alias="STRIX_LLM_TEMPERATURE") + top_p: float | None = Field(default=None, alias="STRIX_LLM_TOP_P") + max_tokens: int | None = Field(default=None, alias="STRIX_LLM_MAX_TOKENS") class RuntimeSettings(BaseSettings): diff --git a/strix/core/inputs.py b/strix/core/inputs.py index b86daa4c..b4aa6775 100644 --- a/strix/core/inputs.py +++ b/strix/core/inputs.py @@ -110,11 +110,17 @@ def make_model_settings( reasoning_effort: ReasoningEffort | None, *, model_name: str, + temperature: float | None = None, + top_p: float | None = None, + max_tokens: int | None = None, ) -> ModelSettings: model_settings = ModelSettings( parallel_tool_calls=False, retry=DEFAULT_MODEL_RETRY, include_usage=True, + temperature=temperature, + top_p=top_p, + max_tokens=max_tokens, ) if ( reasoning_effort is not None diff --git a/strix/core/runner.py b/strix/core/runner.py index a7371e9f..0630a599 100644 --- a/strix/core/runner.py +++ b/strix/core/runner.py @@ -156,6 +156,9 @@ async def run_strix_scan( model_settings = make_model_settings( settings.llm.reasoning_effort, model_name=resolved_model, + temperature=settings.llm.temperature, + top_p=settings.llm.top_p, + max_tokens=settings.llm.max_tokens, ) run_config = RunConfig( model=resolved_model,