feat(llm): expose temperature, top_p, and max_tokens settings

Add optional generation parameters so users can steer model behavior —
particularly useful for local / OpenAI-compatible models that need a
lower temperature for steadier tool calling (see #514).

- STRIX_LLM_TEMPERATURE, STRIX_LLM_TOP_P, STRIX_LLM_MAX_TOKENS
  (all unset by default -> provider defaults, so no behavior change).
- Threaded through make_model_settings into the SDK ModelSettings used
  for the scan. Params a given model rejects are dropped automatically
  (litellm.drop_params is already enabled).
- Documented in docs/advanced/configuration.mdx.

Closes #514
This commit is contained in:
VoidChecksum 2026-06-09 05:07:30 +00:00
parent 3c870c5159
commit eb4bff621a
4 changed files with 24 additions and 0 deletions

View file

@ -31,6 +31,18 @@ Configure Strix using environment variables or a config file.
Control thinking effort for reasoning models. Valid values: `none`, `minimal`, `low`, `medium`, `high`, `xhigh`. Defaults to `medium` for quick scan mode.
</ParamField>
<ParamField path="STRIX_LLM_TEMPERATURE" type="number">
Sampling temperature passed to the model. Unset by default (uses the provider default). Useful for steadier tool use on local/OpenAI-compatible models. Parameters unsupported by a given model are dropped automatically.
</ParamField>
<ParamField path="STRIX_LLM_TOP_P" type="number">
Nucleus sampling `top_p` passed to the model. Unset by default.
</ParamField>
<ParamField path="STRIX_LLM_MAX_TOKENS" type="integer">
Maximum number of tokens to generate per LLM response. Unset by default (uses the provider default).
</ParamField>
<ParamField path="STRIX_MEMORY_COMPRESSOR_TIMEOUT" default="30" type="integer">
Timeout in seconds for memory compression operations (context summarization).
</ParamField>

View file

@ -37,6 +37,9 @@ class LlmSettings(BaseSettings):
)
reasoning_effort: ReasoningEffort = Field(default="high", alias="STRIX_REASONING_EFFORT")
timeout: int = Field(default=300, alias="LLM_TIMEOUT")
temperature: float | None = Field(default=None, alias="STRIX_LLM_TEMPERATURE")
top_p: float | None = Field(default=None, alias="STRIX_LLM_TOP_P")
max_tokens: int | None = Field(default=None, alias="STRIX_LLM_MAX_TOKENS")
class RuntimeSettings(BaseSettings):

View file

@ -110,11 +110,17 @@ def make_model_settings(
reasoning_effort: ReasoningEffort | None,
*,
model_name: str,
temperature: float | None = None,
top_p: float | None = None,
max_tokens: int | None = None,
) -> ModelSettings:
model_settings = ModelSettings(
parallel_tool_calls=False,
retry=DEFAULT_MODEL_RETRY,
include_usage=True,
temperature=temperature,
top_p=top_p,
max_tokens=max_tokens,
)
if (
reasoning_effort is not None

View file

@ -156,6 +156,9 @@ async def run_strix_scan(
model_settings = make_model_settings(
settings.llm.reasoning_effort,
model_name=resolved_model,
temperature=settings.llm.temperature,
top_p=settings.llm.top_p,
max_tokens=settings.llm.max_tokens,
)
run_config = RunConfig(
model=resolved_model,