From 3cd6c93fa0e82c5fe997c846865b6bb91c4f8ecb Mon Sep 17 00:00:00 2001 From: siundu254 Date: Thu, 1 Oct 2026 10:54:20 +0300 Subject: [PATCH] Fail on Severity --- docs/integrations/ci-cd.mdx | 6 + docs/integrations/github-actions.mdx | 2 + docs/usage/cli.mdx | 13 +- .../ci-security-scanning-with-strix/SKILL.md | 2 +- .../penetration-testing-with-strix/SKILL.md | 3 +- strix/interface/cli_args.py | 22 ++++ strix/interface/main.py | 29 ++++- tests/test_cli_fail_on.py | 112 ++++++++++++++++++ 8 files changed, 184 insertions(+), 5 deletions(-) create mode 100644 tests/test_cli_fail_on.py diff --git a/docs/integrations/ci-cd.mdx b/docs/integrations/ci-cd.mdx index f55ebc16..e8361e0e 100644 --- a/docs/integrations/ci-cd.mdx +++ b/docs/integrations/ci-cd.mdx @@ -27,6 +27,12 @@ strix -n --target ./app --scan-mode quick --scope-mode diff --diff-base origin/m | 1 | Execution error | | 2 | Vulnerabilities found | +To fail only on serious findings, pass `--fail-on` with a minimum severity. Lower findings are still reported but exit `0`: + +```bash +strix -n --target ./app --scan-mode quick --fail-on high +``` + ## GitLab CI ```yaml .gitlab-ci.yml diff --git a/docs/integrations/github-actions.mdx b/docs/integrations/github-actions.mdx index 8364c425..4dd3886a 100644 --- a/docs/integrations/github-actions.mdx +++ b/docs/integrations/github-actions.mdx @@ -49,6 +49,8 @@ The workflow fails when vulnerabilities are found: | 0 | Pass — No vulnerabilities | | 2 | Fail — Vulnerabilities found | +Add `--fail-on high` (or `critical`, `medium`, `low`) to fail only on findings at or above that severity. Lower findings still appear in the report. + ## Scan Modes for CI | Mode | Duration | Use Case | diff --git a/docs/usage/cli.mdx b/docs/usage/cli.mdx index 699fb1cb..4e38b31d 100644 --- a/docs/usage/cli.mdx +++ b/docs/usage/cli.mdx @@ -60,6 +60,14 @@ strix (--target | --target-list ) [options] Run in headless mode without TUI. Ideal for CI/CD. + + Headless mode only (requires `-n`). Minimum severity that makes the run exit + `2`: `critical`, `high`, `medium`, `low`, or `info`. Findings below the + threshold are still written to every report artifact. A finding whose + severity Strix does not recognize always counts, so the gate never passes on a + value it cannot rank. Omit the flag to exit `2` on any finding. + + Path to a custom config file (JSON) to use instead of `~/.strix/cli-config.json`. @@ -132,6 +140,9 @@ strix --target api.example.com --instruction "Focus on IDOR and auth bypass" # CI/CD mode strix -n --target ./ --scan-mode quick +# CI/CD mode, failing only on high or critical findings +strix -n --target ./ --scan-mode quick --fail-on high + # Cap cost and per-agent turns strix --target https://example.com --max-budget 25 --max-turns 300 @@ -161,4 +172,4 @@ strix --target https://app.com --workspace-file ./openapi.yaml:specs/openapi.yam |------|---------| | 0 | Scan completed successfully (interactive mode always exits `0`; in headless mode, `0` means no vulnerabilities were found) | | 1 | A fatal error occurred before or during the scan (e.g. missing environment variables, Docker unavailable, invalid config file, diff-scope resolution failure, or an unhandled error) | -| 2 | Vulnerabilities found (headless mode only) | +| 2 | Vulnerabilities found (headless mode only; with `--fail-on`, only findings at or above that severity) | diff --git a/skills/ci-security-scanning-with-strix/SKILL.md b/skills/ci-security-scanning-with-strix/SKILL.md index e53be3ff..41293f63 100644 --- a/skills/ci-security-scanning-with-strix/SKILL.md +++ b/skills/ci-security-scanning-with-strix/SKILL.md @@ -67,7 +67,7 @@ Then tell the user to add two repository secrets: `STRIX_LLM` (model id, for exa Notes: - In CI/headless runs Strix automatically scopes to the PR's changed files (`--scope-mode auto`). If diff resolution fails, keep `fetch-depth: 0` or set `--diff-base` to the PR's actual base branch — use `origin/${{ github.base_ref }}` in GitHub Actions rather than a hard-coded `origin/main`, since repos use different default branches. -- Exit codes: `0` pass, `2` vulnerabilities found (fails the job), `1` setup error. +- Exit codes: `0` pass, `2` vulnerabilities found (fails the job), `1` setup error. Add `--fail-on high` to fail only on high/critical findings; lower ones are still reported. - The runner needs Docker (default GitHub-hosted Ubuntu runners have it). - **Size the budget so the scan completes — do not let it fail open.** A `0` exit means "no validated vulnerabilities in what was analyzed"; if `--max-budget` is hit before the diff is fully covered, the scan wraps up early and can still exit `0`. The "Fail unless the scan completed" step above narrows the gap: `strix_runs//run.json` is `"stopped"` when the scan was cut off at the hard budget limit without a final report. It is not a complete guard — the agents get graduated wrap-up warnings before that limit, and a run that wraps up on a warning still calls `finish_scan` and records `"completed"` with partial coverage. So keep that step in any pipeline that gates merges **and** give the scan real headroom (compare `run.json`'s `llm_usage.cost` against `--max-budget`; if it ran right up to the cap, raise it). For a `quick` diff-scoped PR scan `--max-budget 10` is usually ample, raise it for large diffs. diff --git a/skills/penetration-testing-with-strix/SKILL.md b/skills/penetration-testing-with-strix/SKILL.md index 5f7a38b8..7090d2b5 100644 --- a/skills/penetration-testing-with-strix/SKILL.md +++ b/skills/penetration-testing-with-strix/SKILL.md @@ -94,6 +94,7 @@ Key flags: | `--workspace-file PATH[:DEST]` | Copy a file from this machine into `/workspace` before the scan, for a wordlist, a spec, or notes. Repeatable. | | `--max-budget USD` | Hard LLM spend cap; scan wraps up cleanly at the limit. | | `--max-turns N` | Per-agent turn cap (default 500). | +| `--fail-on SEVERITY` | Headless only: exit `2` only for findings at or above `critical`/`high`/`medium`/`low`/`info`. Default: any finding. | | `--resume RUN_NAME` | Resume a prior run from `strix_runs/`, with its agent history and targets. Cannot be combined with `-t`. | | `--scope-mode` | For code targets: `auto` (diff-scope in CI/headless), `diff` (force changed files only), `full` (whole tree). | | `--diff-base REF` | Branch or commit that `diff` scope compares against. Defaults to the repo's default branch. | @@ -104,7 +105,7 @@ Scans take minutes (`quick`) to hours (`deep`). Run them in the background and p - `0` — finished with no validated vulnerabilities **in what was analyzed** - `1` — fatal error (missing env vars, Docker down, bad config) -- `2` — vulnerabilities found +- `2` — vulnerabilities found (with `--fail-on`, only at or above that severity; lower findings still land in the artifacts) A `0` is not proof of full coverage: if `--max-budget`/`--max-turns` is reached before the scan completes, it wraps up early and still exits `0`. When you need assurance the scan finished, give it enough budget and check `strix_runs//run.json`: a hard budget stop leaves `status: "stopped"`, but an agent that wrapped up early on a budget *warning* still calls `finish_scan` and records `"completed"` — so also sanity-check the run's cost against `--max-budget` and the report's stated coverage before treating a clean result as full coverage. diff --git a/strix/interface/cli_args.py b/strix/interface/cli_args.py index 340a5d47..cea33099 100644 --- a/strix/interface/cli_args.py +++ b/strix/interface/cli_args.py @@ -20,6 +20,10 @@ from strix.interface.utils import ( ) +# Severities ``--fail-on`` accepts, most severe first. +FAIL_ON_SEVERITIES = ("critical", "high", "medium", "low", "info") + + def get_version() -> str: try: from importlib.metadata import version @@ -186,6 +190,21 @@ Strix Cloud: ), ) + parser.add_argument( + "--fail-on", + dest="fail_on", + type=str.lower, + choices=FAIL_ON_SEVERITIES, + default=None, + metavar="SEVERITY", + help=( + "Headless mode only: exit 2 only when a finding is at or above this severity " + "(critical, high, medium, low, info). Lower findings are still written to every " + "report artifact. A finding with an unrecognized severity always counts. " + "Default: any finding exits 2." + ), + ) + parser.add_argument( "-m", "--scan-mode", @@ -318,6 +337,9 @@ Strix Cloud: if args.update: sys.exit(0 if self_update() else 1) + if args.fail_on and not args.non_interactive: + parser.error("--fail-on only applies to headless runs; add -n/--non-interactive.") + if args.instruction and args.instruction_file: parser.error( "Cannot specify both --instruction and --instruction-file. Use one or the other." diff --git a/strix/interface/main.py b/strix/interface/main.py index e58f3562..b86ed141 100644 --- a/strix/interface/main.py +++ b/strix/interface/main.py @@ -8,6 +8,7 @@ import asyncio import contextlib import sys from pathlib import Path +from typing import Any from rich.console import Console from rich.panel import Panel @@ -15,7 +16,7 @@ from rich.text import Text from strix.config import codex, load_settings, persist_current from strix.core.paths import run_dir_for -from strix.interface.cli_args import parse_arguments +from strix.interface.cli_args import FAIL_ON_SEVERITIES, parse_arguments from strix.interface.environment import ( check_docker_installed, pull_docker_image, @@ -315,6 +316,30 @@ def display_completion_message(args: argparse.Namespace, results_path: Path) -> notify_update(console) +def findings_fail_build(reports: list[dict[str, Any]], fail_on: str | None) -> bool: + """Whether headless findings should exit 2 under the ``--fail-on`` threshold. + + With no threshold any finding fails. Otherwise a finding fails when its + severity is at or above the threshold. A severity outside the known scale + fails too, so a gate never passes on a value it cannot rank. ``none`` is a + known level below ``info`` and only fails without a threshold. + """ + if not reports: + return False + if fail_on is None: + return True + threshold = FAIL_ON_SEVERITIES.index(fail_on) + for report in reports: + severity = str(report.get("severity") or "").strip().lower() + if severity == "none": + continue + if severity not in FAIL_ON_SEVERITIES: + return True + if FAIL_ON_SEVERITIES.index(severity) <= threshold: + return True + return False + + def _print_error_panel(title: str, message: str) -> None: console = Console() error_text = Text() @@ -501,7 +526,7 @@ def main() -> None: if args.non_interactive: report_state = get_global_report_state() - if report_state and report_state.vulnerability_reports: + if report_state and findings_fail_build(report_state.vulnerability_reports, args.fail_on): sys.exit(2) diff --git a/tests/test_cli_fail_on.py b/tests/test_cli_fail_on.py new file mode 100644 index 00000000..d0525b14 --- /dev/null +++ b/tests/test_cli_fail_on.py @@ -0,0 +1,112 @@ +"""Tests for the --fail-on headless severity gate.""" + +from __future__ import annotations + +import importlib +import sys +from types import SimpleNamespace +from typing import Any + +import pytest + + +cli_main: Any = importlib.import_module("strix.interface.main") + + +def _stub_settings(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr( + cli_main, + "load_settings", + lambda: SimpleNamespace(runtime=SimpleNamespace(max_local_copy_mb=1024)), + ) + + +def _reports(*severities: str | None) -> list[dict[str, Any]]: + return [{"id": f"vuln-{i:04d}", "severity": sev} for i, sev in enumerate(severities, 1)] + + +def test_no_findings_never_fail() -> None: + assert not cli_main.findings_fail_build([], None) + assert not cli_main.findings_fail_build([], "info") + + +@pytest.mark.parametrize("severity", ["critical", "high", "medium", "low", "info", "none"]) +def test_without_threshold_any_finding_fails(severity: str) -> None: + assert cli_main.findings_fail_build(_reports(severity), None) + + +@pytest.mark.parametrize( + ("fail_on", "severity", "expected"), + [ + ("high", "critical", True), + ("high", "high", True), + ("high", "medium", False), + ("high", "low", False), + ("high", "info", False), + ("critical", "high", False), + ("critical", "critical", True), + ("info", "info", True), + ("info", "low", True), + ("medium", "HIGH", True), + ("medium", " Low ", False), + ], +) +def test_threshold_compares_severity(fail_on: str, severity: str, expected: bool) -> None: + assert cli_main.findings_fail_build(_reports(severity), fail_on) is expected + + +def test_one_finding_at_threshold_fails_a_mixed_run() -> None: + assert cli_main.findings_fail_build(_reports("info", "low", "high"), "high") + + +@pytest.mark.parametrize("severity", ["severe", "", None]) +def test_unrecognized_severity_fails_closed(severity: str | None) -> None: + assert cli_main.findings_fail_build(_reports(severity), "critical") + + +def test_none_severity_passes_any_threshold() -> None: + assert not cli_main.findings_fail_build(_reports("none"), "info") + + +def test_parse_fail_on_is_case_insensitive(monkeypatch: pytest.MonkeyPatch) -> None: + _stub_settings(monkeypatch) + monkeypatch.setattr( + sys, "argv", ["strix", "-t", "https://test.com/", "-n", "--fail-on", "HIGH"] + ) + + args = cli_main.parse_arguments() + + assert args.fail_on == "high" + + +def test_parse_fail_on_defaults_to_none(monkeypatch: pytest.MonkeyPatch) -> None: + _stub_settings(monkeypatch) + monkeypatch.setattr(sys, "argv", ["strix", "-t", "https://test.com/", "-n"]) + + assert cli_main.parse_arguments().fail_on is None + + +def test_parse_fail_on_rejects_unknown_severity( + monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str] +) -> None: + _stub_settings(monkeypatch) + monkeypatch.setattr( + sys, "argv", ["strix", "-t", "https://test.com/", "-n", "--fail-on", "severe"] + ) + + with pytest.raises(SystemExit): + cli_main.parse_arguments() + + assert "--fail-on" in capsys.readouterr().err + + +def test_parse_fail_on_requires_non_interactive( + monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str] +) -> None: + _stub_settings(monkeypatch) + monkeypatch.setattr(sys, "argv", ["strix", "-t", "https://test.com/", "--fail-on", "high"]) + + with pytest.raises(SystemExit): + cli_main.parse_arguments() + + assert "--fail-on only applies to headless runs" in capsys.readouterr().err