diff --git a/docs/advanced/configuration.mdx b/docs/advanced/configuration.mdx
index 9ab7f017..e85e4716 100644
--- a/docs/advanced/configuration.mdx
+++ b/docs/advanced/configuration.mdx
@@ -31,6 +31,18 @@ Configure Strix using environment variables or a config file.
Control thinking effort for reasoning models. Valid values: `none`, `minimal`, `low`, `medium`, `high`, `xhigh`. Defaults to `medium` for quick scan mode.
+
+ Sampling temperature passed to the model. Unset by default (uses the provider default). Useful for steadier tool use on local/OpenAI-compatible models. Parameters unsupported by a given model are dropped automatically.
+
+
+
+ Nucleus sampling `top_p` passed to the model. Unset by default.
+
+
+
+ Maximum number of tokens to generate per LLM response. Unset by default (uses the provider default).
+
+
Timeout in seconds for memory compression operations (context summarization).
diff --git a/strix/config/settings.py b/strix/config/settings.py
index 1458e1ff..578c3f44 100644
--- a/strix/config/settings.py
+++ b/strix/config/settings.py
@@ -37,6 +37,9 @@ class LlmSettings(BaseSettings):
)
reasoning_effort: ReasoningEffort = Field(default="high", alias="STRIX_REASONING_EFFORT")
timeout: int = Field(default=300, alias="LLM_TIMEOUT")
+ temperature: float | None = Field(default=None, alias="STRIX_LLM_TEMPERATURE")
+ top_p: float | None = Field(default=None, alias="STRIX_LLM_TOP_P")
+ max_tokens: int | None = Field(default=None, alias="STRIX_LLM_MAX_TOKENS")
class RuntimeSettings(BaseSettings):
diff --git a/strix/core/inputs.py b/strix/core/inputs.py
index b86daa4c..b4aa6775 100644
--- a/strix/core/inputs.py
+++ b/strix/core/inputs.py
@@ -110,11 +110,17 @@ def make_model_settings(
reasoning_effort: ReasoningEffort | None,
*,
model_name: str,
+ temperature: float | None = None,
+ top_p: float | None = None,
+ max_tokens: int | None = None,
) -> ModelSettings:
model_settings = ModelSettings(
parallel_tool_calls=False,
retry=DEFAULT_MODEL_RETRY,
include_usage=True,
+ temperature=temperature,
+ top_p=top_p,
+ max_tokens=max_tokens,
)
if (
reasoning_effort is not None
diff --git a/strix/core/runner.py b/strix/core/runner.py
index a7371e9f..0630a599 100644
--- a/strix/core/runner.py
+++ b/strix/core/runner.py
@@ -156,6 +156,9 @@ async def run_strix_scan(
model_settings = make_model_settings(
settings.llm.reasoning_effort,
model_name=resolved_model,
+ temperature=settings.llm.temperature,
+ top_p=settings.llm.top_p,
+ max_tokens=settings.llm.max_tokens,
)
run_config = RunConfig(
model=resolved_model,