diff --git a/docs/advanced/configuration.mdx b/docs/advanced/configuration.mdx index 9ab7f017..e85e4716 100644 --- a/docs/advanced/configuration.mdx +++ b/docs/advanced/configuration.mdx @@ -31,6 +31,18 @@ Configure Strix using environment variables or a config file. Control thinking effort for reasoning models. Valid values: `none`, `minimal`, `low`, `medium`, `high`, `xhigh`. Defaults to `medium` for quick scan mode. + + Sampling temperature passed to the model. Unset by default (uses the provider default). Useful for steadier tool use on local/OpenAI-compatible models. Parameters unsupported by a given model are dropped automatically. + + + + Nucleus sampling `top_p` passed to the model. Unset by default. + + + + Maximum number of tokens to generate per LLM response. Unset by default (uses the provider default). + + Timeout in seconds for memory compression operations (context summarization). diff --git a/strix/config/settings.py b/strix/config/settings.py index 1458e1ff..578c3f44 100644 --- a/strix/config/settings.py +++ b/strix/config/settings.py @@ -37,6 +37,9 @@ class LlmSettings(BaseSettings): ) reasoning_effort: ReasoningEffort = Field(default="high", alias="STRIX_REASONING_EFFORT") timeout: int = Field(default=300, alias="LLM_TIMEOUT") + temperature: float | None = Field(default=None, alias="STRIX_LLM_TEMPERATURE") + top_p: float | None = Field(default=None, alias="STRIX_LLM_TOP_P") + max_tokens: int | None = Field(default=None, alias="STRIX_LLM_MAX_TOKENS") class RuntimeSettings(BaseSettings): diff --git a/strix/core/inputs.py b/strix/core/inputs.py index b86daa4c..b4aa6775 100644 --- a/strix/core/inputs.py +++ b/strix/core/inputs.py @@ -110,11 +110,17 @@ def make_model_settings( reasoning_effort: ReasoningEffort | None, *, model_name: str, + temperature: float | None = None, + top_p: float | None = None, + max_tokens: int | None = None, ) -> ModelSettings: model_settings = ModelSettings( parallel_tool_calls=False, retry=DEFAULT_MODEL_RETRY, include_usage=True, + temperature=temperature, + top_p=top_p, + max_tokens=max_tokens, ) if ( reasoning_effort is not None diff --git a/strix/core/runner.py b/strix/core/runner.py index a7371e9f..0630a599 100644 --- a/strix/core/runner.py +++ b/strix/core/runner.py @@ -156,6 +156,9 @@ async def run_strix_scan( model_settings = make_model_settings( settings.llm.reasoning_effort, model_name=resolved_model, + temperature=settings.llm.temperature, + top_p=settings.llm.top_p, + max_tokens=settings.llm.max_tokens, ) run_config = RunConfig( model=resolved_model,