Fail on Severity

This commit is contained in:
siundu254 2026-10-01 10:54:20 +03:00 • committed by Ahmed Allam
parent 65172fecd8
commit 3cd6c93fa0
8 changed files with 184 additions and 5 deletions

View file

@ -27,6 +27,12 @@ strix -n --target ./app --scan-mode quick --scope-mode diff --diff-base origin/m
| 1 | Execution error | | 1 | Execution error |
| 2 | Vulnerabilities found | | 2 | Vulnerabilities found |
To fail only on serious findings, pass `--fail-on` with a minimum severity. Lower findings are still reported but exit `0`:
```bash
strix -n --target ./app --scan-mode quick --fail-on high
```
## GitLab CI ## GitLab CI
```yaml .gitlab-ci.yml ```yaml .gitlab-ci.yml

View file

@ -49,6 +49,8 @@ The workflow fails when vulnerabilities are found:
| 0 | Pass — No vulnerabilities | | 0 | Pass — No vulnerabilities |
| 2 | Fail — Vulnerabilities found | | 2 | Fail — Vulnerabilities found |
Add `--fail-on high` (or `critical`, `medium`, `low`) to fail only on findings at or above that severity. Lower findings still appear in the report.
## Scan Modes for CI ## Scan Modes for CI
| Mode | Duration | Use Case | | Mode | Duration | Use Case |

View file

@ -60,6 +60,14 @@ strix (--target <target> | --target-list <path>) [options]
Run in headless mode without TUI. Ideal for CI/CD. Run in headless mode without TUI. Ideal for CI/CD.
</ParamField> </ParamField>
<ParamField path="--fail-on" type="string">
Headless mode only (requires `-n`). Minimum severity that makes the run exit
`2`: `critical`, `high`, `medium`, `low`, or `info`. Findings below the
threshold are still written to every report artifact. A finding whose
severity Strix does not recognize always counts, so the gate never passes on a
value it cannot rank. Omit the flag to exit `2` on any finding.
</ParamField>
<ParamField path="--config" type="string"> <ParamField path="--config" type="string">
Path to a custom config file (JSON) to use instead of `~/.strix/cli-config.json`. Path to a custom config file (JSON) to use instead of `~/.strix/cli-config.json`.
</ParamField> </ParamField>
@ -132,6 +140,9 @@ strix --target api.example.com --instruction "Focus on IDOR and auth bypass"
# CI/CD mode # CI/CD mode
strix -n --target ./ --scan-mode quick strix -n --target ./ --scan-mode quick
# CI/CD mode, failing only on high or critical findings
strix -n --target ./ --scan-mode quick --fail-on high
# Cap cost and per-agent turns # Cap cost and per-agent turns
strix --target https://example.com --max-budget 25 --max-turns 300 strix --target https://example.com --max-budget 25 --max-turns 300
@ -161,4 +172,4 @@ strix --target https://app.com --workspace-file ./openapi.yaml:specs/openapi.yam
|------|---------| |------|---------|
| 0 | Scan completed successfully (interactive mode always exits `0`; in headless mode, `0` means no vulnerabilities were found) | | 0 | Scan completed successfully (interactive mode always exits `0`; in headless mode, `0` means no vulnerabilities were found) |
| 1 | A fatal error occurred before or during the scan (e.g. missing environment variables, Docker unavailable, invalid config file, diff-scope resolution failure, or an unhandled error) | | 1 | A fatal error occurred before or during the scan (e.g. missing environment variables, Docker unavailable, invalid config file, diff-scope resolution failure, or an unhandled error) |
| 2 | Vulnerabilities found (headless mode only) | | 2 | Vulnerabilities found (headless mode only; with `--fail-on`, only findings at or above that severity) |

View file

@ -67,7 +67,7 @@ Then tell the user to add two repository secrets: `STRIX_LLM` (model id, for exa
Notes: Notes:
- In CI/headless runs Strix automatically scopes to the PR's changed files (`--scope-mode auto`). If diff resolution fails, keep `fetch-depth: 0` or set `--diff-base` to the PR's actual base branch — use `origin/${{ github.base_ref }}` in GitHub Actions rather than a hard-coded `origin/main`, since repos use different default branches. - In CI/headless runs Strix automatically scopes to the PR's changed files (`--scope-mode auto`). If diff resolution fails, keep `fetch-depth: 0` or set `--diff-base` to the PR's actual base branch — use `origin/${{ github.base_ref }}` in GitHub Actions rather than a hard-coded `origin/main`, since repos use different default branches.
- Exit codes: `0` pass, `2` vulnerabilities found (fails the job), `1` setup error. - Exit codes: `0` pass, `2` vulnerabilities found (fails the job), `1` setup error. Add `--fail-on high` to fail only on high/critical findings; lower ones are still reported.
- The runner needs Docker (default GitHub-hosted Ubuntu runners have it). - The runner needs Docker (default GitHub-hosted Ubuntu runners have it).
- **Size the budget so the scan completes — do not let it fail open.** A `0` exit means "no validated vulnerabilities in what was analyzed"; if `--max-budget` is hit before the diff is fully covered, the scan wraps up early and can still exit `0`. The "Fail unless the scan completed" step above narrows the gap: `strix_runs/<run>/run.json` is `"stopped"` when the scan was cut off at the hard budget limit without a final report. It is not a complete guard — the agents get graduated wrap-up warnings before that limit, and a run that wraps up on a warning still calls `finish_scan` and records `"completed"` with partial coverage. So keep that step in any pipeline that gates merges **and** give the scan real headroom (compare `run.json`'s `llm_usage.cost` against `--max-budget`; if it ran right up to the cap, raise it). For a `quick` diff-scoped PR scan `--max-budget 10` is usually ample, raise it for large diffs. - **Size the budget so the scan completes — do not let it fail open.** A `0` exit means "no validated vulnerabilities in what was analyzed"; if `--max-budget` is hit before the diff is fully covered, the scan wraps up early and can still exit `0`. The "Fail unless the scan completed" step above narrows the gap: `strix_runs/<run>/run.json` is `"stopped"` when the scan was cut off at the hard budget limit without a final report. It is not a complete guard — the agents get graduated wrap-up warnings before that limit, and a run that wraps up on a warning still calls `finish_scan` and records `"completed"` with partial coverage. So keep that step in any pipeline that gates merges **and** give the scan real headroom (compare `run.json`'s `llm_usage.cost` against `--max-budget`; if it ran right up to the cap, raise it). For a `quick` diff-scoped PR scan `--max-budget 10` is usually ample, raise it for large diffs.

View file

@ -94,6 +94,7 @@ Key flags:
| `--workspace-file PATH[:DEST]` | Copy a file from this machine into `/workspace` before the scan, for a wordlist, a spec, or notes. Repeatable. | | `--workspace-file PATH[:DEST]` | Copy a file from this machine into `/workspace` before the scan, for a wordlist, a spec, or notes. Repeatable. |
| `--max-budget USD` | Hard LLM spend cap; scan wraps up cleanly at the limit. | | `--max-budget USD` | Hard LLM spend cap; scan wraps up cleanly at the limit. |
| `--max-turns N` | Per-agent turn cap (default 500). | | `--max-turns N` | Per-agent turn cap (default 500). |
| `--fail-on SEVERITY` | Headless only: exit `2` only for findings at or above `critical`/`high`/`medium`/`low`/`info`. Default: any finding. |
| `--resume RUN_NAME` | Resume a prior run from `strix_runs/`, with its agent history and targets. Cannot be combined with `-t`. | | `--resume RUN_NAME` | Resume a prior run from `strix_runs/`, with its agent history and targets. Cannot be combined with `-t`. |
| `--scope-mode` | For code targets: `auto` (diff-scope in CI/headless), `diff` (force changed files only), `full` (whole tree). | | `--scope-mode` | For code targets: `auto` (diff-scope in CI/headless), `diff` (force changed files only), `full` (whole tree). |
| `--diff-base REF` | Branch or commit that `diff` scope compares against. Defaults to the repo's default branch. | | `--diff-base REF` | Branch or commit that `diff` scope compares against. Defaults to the repo's default branch. |
@ -104,7 +105,7 @@ Scans take minutes (`quick`) to hours (`deep`). Run them in the background and p
- `0` — finished with no validated vulnerabilities **in what was analyzed** - `0` — finished with no validated vulnerabilities **in what was analyzed**
- `1` — fatal error (missing env vars, Docker down, bad config) - `1` — fatal error (missing env vars, Docker down, bad config)
- `2` — vulnerabilities found - `2` — vulnerabilities found (with `--fail-on`, only at or above that severity; lower findings still land in the artifacts)
A `0` is not proof of full coverage: if `--max-budget`/`--max-turns` is reached before the scan completes, it wraps up early and still exits `0`. When you need assurance the scan finished, give it enough budget and check `strix_runs/<run>/run.json`: a hard budget stop leaves `status: "stopped"`, but an agent that wrapped up early on a budget *warning* still calls `finish_scan` and records `"completed"` — so also sanity-check the run's cost against `--max-budget` and the report's stated coverage before treating a clean result as full coverage. A `0` is not proof of full coverage: if `--max-budget`/`--max-turns` is reached before the scan completes, it wraps up early and still exits `0`. When you need assurance the scan finished, give it enough budget and check `strix_runs/<run>/run.json`: a hard budget stop leaves `status: "stopped"`, but an agent that wrapped up early on a budget *warning* still calls `finish_scan` and records `"completed"` — so also sanity-check the run's cost against `--max-budget` and the report's stated coverage before treating a clean result as full coverage.

View file

@ -20,6 +20,10 @@ from strix.interface.utils import (
) )
# Severities ``--fail-on`` accepts, most severe first.
FAIL_ON_SEVERITIES = ("critical", "high", "medium", "low", "info")
def get_version() -> str: def get_version() -> str:
try: try:
from importlib.metadata import version from importlib.metadata import version
@ -186,6 +190,21 @@ Strix Cloud:
), ),
) )
parser.add_argument(
"--fail-on",
dest="fail_on",
type=str.lower,
choices=FAIL_ON_SEVERITIES,
default=None,
metavar="SEVERITY",
help=(
"Headless mode only: exit 2 only when a finding is at or above this severity "
"(critical, high, medium, low, info). Lower findings are still written to every "
"report artifact. A finding with an unrecognized severity always counts. "
"Default: any finding exits 2."
),
)
parser.add_argument( parser.add_argument(
"-m", "-m",
"--scan-mode", "--scan-mode",
@ -318,6 +337,9 @@ Strix Cloud:
if args.update: if args.update:
sys.exit(0 if self_update() else 1) sys.exit(0 if self_update() else 1)
if args.fail_on and not args.non_interactive:
parser.error("--fail-on only applies to headless runs; add -n/--non-interactive.")
if args.instruction and args.instruction_file: if args.instruction and args.instruction_file:
parser.error( parser.error(
"Cannot specify both --instruction and --instruction-file. Use one or the other." "Cannot specify both --instruction and --instruction-file. Use one or the other."

View file

@ -8,6 +8,7 @@ import asyncio
import contextlib import contextlib
import sys import sys
from pathlib import Path from pathlib import Path
from typing import Any
from rich.console import Console from rich.console import Console
from rich.panel import Panel from rich.panel import Panel
@ -15,7 +16,7 @@ from rich.text import Text
from strix.config import codex, load_settings, persist_current from strix.config import codex, load_settings, persist_current
from strix.core.paths import run_dir_for from strix.core.paths import run_dir_for
from strix.interface.cli_args import parse_arguments from strix.interface.cli_args import FAIL_ON_SEVERITIES, parse_arguments
from strix.interface.environment import ( from strix.interface.environment import (
check_docker_installed, check_docker_installed,
pull_docker_image, pull_docker_image,
@ -315,6 +316,30 @@ def display_completion_message(args: argparse.Namespace, results_path: Path) ->
notify_update(console) notify_update(console)
def findings_fail_build(reports: list[dict[str, Any]], fail_on: str | None) -> bool:
"""Whether headless findings should exit 2 under the ``--fail-on`` threshold.
With no threshold any finding fails. Otherwise a finding fails when its
severity is at or above the threshold. A severity outside the known scale
fails too, so a gate never passes on a value it cannot rank. ``none`` is a
known level below ``info`` and only fails without a threshold.
"""
if not reports:
return False
if fail_on is None:
return True
threshold = FAIL_ON_SEVERITIES.index(fail_on)
for report in reports:
severity = str(report.get("severity") or "").strip().lower()
if severity == "none":
continue
if severity not in FAIL_ON_SEVERITIES:
return True
if FAIL_ON_SEVERITIES.index(severity) <= threshold:
return True
return False
def _print_error_panel(title: str, message: str) -> None: def _print_error_panel(title: str, message: str) -> None:
console = Console() console = Console()
error_text = Text() error_text = Text()
@ -501,7 +526,7 @@ def main() -> None:
if args.non_interactive: if args.non_interactive:
report_state = get_global_report_state() report_state = get_global_report_state()
if report_state and report_state.vulnerability_reports: if report_state and findings_fail_build(report_state.vulnerability_reports, args.fail_on):
sys.exit(2) sys.exit(2)

112
tests/test_cli_fail_on.py Normal file
View file

@ -0,0 +1,112 @@
"""Tests for the --fail-on headless severity gate."""
from __future__ import annotations
import importlib
import sys
from types import SimpleNamespace
from typing import Any
import pytest
cli_main: Any = importlib.import_module("strix.interface.main")
def _stub_settings(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setattr(
cli_main,
"load_settings",
lambda: SimpleNamespace(runtime=SimpleNamespace(max_local_copy_mb=1024)),
)
def _reports(*severities: str | None) -> list[dict[str, Any]]:
return [{"id": f"vuln-{i:04d}", "severity": sev} for i, sev in enumerate(severities, 1)]
def test_no_findings_never_fail() -> None:
assert not cli_main.findings_fail_build([], None)
assert not cli_main.findings_fail_build([], "info")
@pytest.mark.parametrize("severity", ["critical", "high", "medium", "low", "info", "none"])
def test_without_threshold_any_finding_fails(severity: str) -> None:
assert cli_main.findings_fail_build(_reports(severity), None)
@pytest.mark.parametrize(
("fail_on", "severity", "expected"),
[
("high", "critical", True),
("high", "high", True),
("high", "medium", False),
("high", "low", False),
("high", "info", False),
("critical", "high", False),
("critical", "critical", True),
("info", "info", True),
("info", "low", True),
("medium", "HIGH", True),
("medium", " Low ", False),
],
)
def test_threshold_compares_severity(fail_on: str, severity: str, expected: bool) -> None:
assert cli_main.findings_fail_build(_reports(severity), fail_on) is expected
def test_one_finding_at_threshold_fails_a_mixed_run() -> None:
assert cli_main.findings_fail_build(_reports("info", "low", "high"), "high")
@pytest.mark.parametrize("severity", ["severe", "", None])
def test_unrecognized_severity_fails_closed(severity: str | None) -> None:
assert cli_main.findings_fail_build(_reports(severity), "critical")
def test_none_severity_passes_any_threshold() -> None:
assert not cli_main.findings_fail_build(_reports("none"), "info")
def test_parse_fail_on_is_case_insensitive(monkeypatch: pytest.MonkeyPatch) -> None:
_stub_settings(monkeypatch)
monkeypatch.setattr(
sys, "argv", ["strix", "-t", "https://test.com/", "-n", "--fail-on", "HIGH"]
)
args = cli_main.parse_arguments()
assert args.fail_on == "high"
def test_parse_fail_on_defaults_to_none(monkeypatch: pytest.MonkeyPatch) -> None:
_stub_settings(monkeypatch)
monkeypatch.setattr(sys, "argv", ["strix", "-t", "https://test.com/", "-n"])
assert cli_main.parse_arguments().fail_on is None
def test_parse_fail_on_rejects_unknown_severity(
monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str]
) -> None:
_stub_settings(monkeypatch)
monkeypatch.setattr(
sys, "argv", ["strix", "-t", "https://test.com/", "-n", "--fail-on", "severe"]
)
with pytest.raises(SystemExit):
cli_main.parse_arguments()
assert "--fail-on" in capsys.readouterr().err
def test_parse_fail_on_requires_non_interactive(
monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str]
) -> None:
_stub_settings(monkeypatch)
monkeypatch.setattr(sys, "argv", ["strix", "-t", "https://test.com/", "--fail-on", "high"])
with pytest.raises(SystemExit):
cli_main.parse_arguments()
assert "--fail-on only applies to headless runs" in capsys.readouterr().err