Address review feedback on #549

- reporting: gate code_locations on whether source is in scope
  (local_code OR repository), not is_whitebox. is_whitebox is False for
  repository targets even though their source is cloned to /workspace,
  so the prior guard silently dropped valid code_locations on repo
  scans. runner now threads source_in_scope; create_vulnerability_report
  keys off it.
- llm timeout: only override litellm.request_timeout when LLM_TIMEOUT is
  explicitly set (llm.model_fields_set), so users who never set it keep
  LiteLLM's ~6000s default instead of being silently capped at the 300s
  default.
- settings: bound STRIX_LLM_TEMPERATURE to [0,2], STRIX_LLM_TOP_P to
  [0,1], STRIX_LLM_MAX_TOKENS to >=1 so invalid values fail at config
  load instead of at the provider; docs note the ranges.
This commit is contained in:
VoidChecksum 2026-06-09 08:17:36 +00:00
parent cdad5dd0a8
commit 3804421322
5 changed files with 21 additions and 18 deletions

View file

@ -32,11 +32,11 @@ Configure Strix using environment variables or a config file.
</ParamField>
<ParamField path="STRIX_LLM_TEMPERATURE" type="number">
Sampling temperature passed to the model. Unset by default (uses the provider default). Useful for steadier tool use on local/OpenAI-compatible models. Parameters unsupported by a given model are dropped automatically.
Sampling temperature (`0.0`–`2.0`) passed to the model. Unset by default (uses the provider default). Useful for steadier tool use on local/OpenAI-compatible models. Parameters unsupported by a given model are dropped automatically.
</ParamField>
<ParamField path="STRIX_LLM_TOP_P" type="number">
Nucleus sampling `top_p` passed to the model. Unset by default.
Nucleus sampling `top_p` (`0.0`–`1.0`) passed to the model. Unset by default.
</ParamField>
<ParamField path="STRIX_LLM_MAX_TOKENS" type="integer">

View file

@ -63,7 +63,8 @@ def configure_sdk_model_defaults(settings: Settings) -> None:
llm = settings.llm
set_tracing_disabled(True)
_configure_litellm_compatibility()
_configure_litellm_request_timeout(llm.timeout)
if "timeout" in llm.model_fields_set:
_configure_litellm_request_timeout(llm.timeout)
if llm.api_key:
set_default_openai_key(llm.api_key, use_for_tracing=False)
_configure_litellm_default("api_key", llm.api_key)
@ -133,11 +134,11 @@ def _configure_litellm_request_timeout(timeout: int) -> None:
"""Apply the configured ``LLM_TIMEOUT`` to LiteLLM-routed scan calls.
The SDK's LiteLLM model invokes ``litellm.acompletion`` without an explicit
per-request timeout, so without this it falls back to LiteLLM's module
default and the documented ``LLM_TIMEOUT`` only affected the warm-up call
(``strix/interface/main.py``). Setting the module-level default makes
``export LLM_TIMEOUT=600`` take effect for the actual scan, restoring the
documented behavior for slow local / self-hosted models.
per-request timeout, so before this the documented ``LLM_TIMEOUT`` only
affected the warm-up call (``strix/interface/main.py``) and scan calls fell
back to LiteLLM's ~6000s module default. The caller applies this only when
``LLM_TIMEOUT`` is explicitly set, so users who never set it keep LiteLLM's
default instead of being silently capped at the 300s ``LLM_TIMEOUT`` default.
"""
import litellm

View file

@ -37,9 +37,9 @@ class LlmSettings(BaseSettings):
)
reasoning_effort: ReasoningEffort = Field(default="high", alias="STRIX_REASONING_EFFORT")
timeout: int = Field(default=300, alias="LLM_TIMEOUT")
temperature: float | None = Field(default=None, alias="STRIX_LLM_TEMPERATURE")
top_p: float | None = Field(default=None, alias="STRIX_LLM_TOP_P")
max_tokens: int | None = Field(default=None, alias="STRIX_LLM_MAX_TOKENS")
temperature: float | None = Field(default=None, alias="STRIX_LLM_TEMPERATURE", ge=0.0, le=2.0)
top_p: float | None = Field(default=None, alias="STRIX_LLM_TOP_P", ge=0.0, le=1.0)
max_tokens: int | None = Field(default=None, alias="STRIX_LLM_MAX_TOKENS", ge=1)
class RuntimeSettings(BaseSettings):

View file

@ -151,6 +151,7 @@ async def run_strix_scan(
targets = scan_config.get("targets") or []
scan_mode = str(scan_config.get("scan_mode") or "deep")
is_whitebox = any(t.get("type") == "local_code" for t in targets)
source_in_scope = is_whitebox or any(t.get("type") == "repository" for t in targets)
skills = list(scan_config.get("skills") or [])
root_task = build_root_task(scan_config)
model_settings = make_model_settings(
@ -220,7 +221,7 @@ async def run_strix_scan(
"agent_id": root_id,
"parent_id": None,
"interactive": interactive,
"is_whitebox": is_whitebox,
"source_in_scope": source_in_scope,
"spawn_child_agent": spawn_child_agent,
}

View file

@ -489,12 +489,13 @@ async def create_vulnerability_report(
- Duplicating the same change across multiple locations.
"""
inner = ctx.context if isinstance(ctx.context, dict) else {}
if not inner.get("is_whitebox") and code_locations:
# Black-box scan: no source tree is available, so any file paths /
# line numbers / snippets in code_locations can only be fabricated.
# Drop them so a hallucinated "Code Analysis" section can never reach
# the customer-facing report (#321).
logger.info("Black-box scan: dropping code_locations from report %r", title)
if not inner.get("source_in_scope") and code_locations:
# No source tree is in scope (e.g. a URL/IP/domain black-box scan), so any
# file paths / line numbers / snippets in code_locations can only be
# fabricated. Drop them so a hallucinated "Code Analysis" section can never
# reach the customer-facing report (#321). Repository and local-code scans
# keep code_locations because the agent can actually read the source.
logger.info("No source in scope: dropping code_locations from report %r", title)
code_locations = None
raw_agent_id = inner.get("agent_id")