mirror of
https://github.com/usestrix/strix.git
synced 2026-10-02 02:13:43 +00:00
677 lines
21 KiB
Python
677 lines
21 KiB
Python
"""Tests for fix candidate anchoring and repository preparation."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import json
|
|
import subprocess
|
|
import sys
|
|
from typing import TYPE_CHECKING
|
|
|
|
import pytest
|
|
|
|
from strix.fix.contracts import (
|
|
BlockerKind,
|
|
CandidateLocation,
|
|
CheckResult,
|
|
CheckStatus,
|
|
CommandSpec,
|
|
FixCandidateV1,
|
|
FixEdit,
|
|
FixPreparationRequestV1,
|
|
PreparationState,
|
|
RegressionTestResult,
|
|
RepairOutcome,
|
|
RepairStatus,
|
|
RepositoryTestPlan,
|
|
ReproductionSpec,
|
|
SourceIdentity,
|
|
SourceIdentityKind,
|
|
VerificationDecision,
|
|
VerificationTarget,
|
|
VerifierResult,
|
|
candidate_from_legacy_report,
|
|
)
|
|
from strix.fix.evidence import command_status, passed_test_count, record_test_execution
|
|
from strix.fix.locations import AnchorStatus, anchor_location
|
|
from strix.fix.prepare import (
|
|
PreparationContext,
|
|
_network_isolation_prefix,
|
|
build_git_manifest,
|
|
build_git_patch,
|
|
prepare_fix,
|
|
run_command,
|
|
)
|
|
|
|
|
|
if TYPE_CHECKING:
|
|
from pathlib import Path
|
|
|
|
|
|
def _regression() -> RegressionTestResult:
|
|
base = CheckResult(
|
|
name="authorization",
|
|
argv=["python", "regression.py"],
|
|
status=CheckStatus.FAILED,
|
|
exit_code=1,
|
|
duration_seconds=0,
|
|
target=VerificationTarget.BASE,
|
|
output="unauthorized access allowed",
|
|
)
|
|
passed = base.model_copy(
|
|
update={
|
|
"status": CheckStatus.PASSED,
|
|
"exit_code": 0,
|
|
"target": VerificationTarget.PATCHED,
|
|
"output": "passed",
|
|
}
|
|
)
|
|
return RegressionTestResult(
|
|
name="authorization",
|
|
expected_base_failure="unauthorized access allowed",
|
|
harness_sha256="a" * 64,
|
|
base=base,
|
|
patched=passed,
|
|
behavior=passed,
|
|
)
|
|
|
|
|
|
def _git(workspace: Path, *args: str) -> str:
|
|
return subprocess.run( # noqa: S603
|
|
["/usr/bin/git", *args],
|
|
cwd=workspace,
|
|
check=True,
|
|
capture_output=True,
|
|
text=True,
|
|
).stdout.strip()
|
|
|
|
|
|
def _workspace(tmp_path: Path) -> tuple[Path, str]:
|
|
workspace = tmp_path / "repo"
|
|
workspace.mkdir()
|
|
_git(workspace, "init")
|
|
_git(workspace, "config", "user.email", "test@example.com")
|
|
_git(workspace, "config", "user.name", "Test")
|
|
(workspace / "app.py").write_text("def result():\n return 'unsafe'\n", encoding="utf-8")
|
|
_git(workspace, "add", "app.py")
|
|
_git(workspace, "commit", "-m", "initial")
|
|
return workspace, _git(workspace, "rev-parse", "HEAD")
|
|
|
|
|
|
def _candidate(
|
|
commit: str,
|
|
*,
|
|
reproduction: ReproductionSpec | None = None,
|
|
) -> FixCandidateV1:
|
|
if reproduction is None:
|
|
reproduction = ReproductionSpec(
|
|
instructions="Confirm that result returns safe.",
|
|
command=CommandSpec(
|
|
name="security reproduction",
|
|
argv=[
|
|
sys.executable,
|
|
"-c",
|
|
"from app import result; assert result() == 'safe'",
|
|
],
|
|
),
|
|
)
|
|
return FixCandidateV1(
|
|
source_identity=SourceIdentity(kind=SourceIdentityKind.COMMIT, value=commit),
|
|
security_invariant="Return a safe value.",
|
|
finding_locations=[
|
|
CandidateLocation(
|
|
file="app.py",
|
|
start_line=1,
|
|
end_line=2,
|
|
snippet="def result():\n return 'unsafe'",
|
|
)
|
|
],
|
|
draft_edits=[
|
|
FixEdit(
|
|
file="app.py",
|
|
start_line=2,
|
|
end_line=2,
|
|
before=" return 'unsafe'",
|
|
after=" return 'safe'",
|
|
)
|
|
],
|
|
reproduction=reproduction,
|
|
)
|
|
|
|
|
|
def _request(candidate: FixCandidateV1, *, attempts: int = 2) -> FixPreparationRequestV1:
|
|
return FixPreparationRequestV1(
|
|
scan_id="scan-1",
|
|
finding_id="finding-1",
|
|
candidate=candidate,
|
|
checks=[
|
|
CommandSpec(
|
|
name="compile",
|
|
argv=[
|
|
sys.executable,
|
|
"-c",
|
|
"compile(open('app.py', encoding='utf-8').read(), 'app.py', 'exec')",
|
|
],
|
|
)
|
|
],
|
|
max_repair_attempts=attempts,
|
|
network_allowed=True,
|
|
)
|
|
|
|
|
|
async def _noop_repair(
|
|
context: PreparationContext,
|
|
_checks: list[CheckResult],
|
|
) -> RepairOutcome:
|
|
path = context.workspace / "app.py"
|
|
if path.exists():
|
|
path.write_text(
|
|
path.read_text(encoding="utf-8").replace("'unsafe'", "'safe'"),
|
|
encoding="utf-8",
|
|
)
|
|
(context.workspace / "test_app.py").write_text(
|
|
"import unittest\nfrom app import result\nclass SecurityTest(unittest.TestCase):\n"
|
|
" def test_safe(self): self.assertEqual(result(), 'safe')\n"
|
|
)
|
|
return RepairOutcome(
|
|
test_plan=RepositoryTestPlan(
|
|
regression_files=["test_app.py"],
|
|
regression_test=CommandSpec(
|
|
name="regression", argv=[sys.executable, "-m", "unittest", "test_app"]
|
|
),
|
|
no_unit_tests_reason="The fixture has no prior unit suite.",
|
|
),
|
|
status=RepairStatus.COMPLETE,
|
|
summary="The repository fix is ready for independent evaluation.",
|
|
)
|
|
|
|
|
|
async def _verified(
|
|
_context: PreparationContext,
|
|
_checks: list[CheckResult],
|
|
) -> VerifierResult:
|
|
return VerifierResult(
|
|
decision=VerificationDecision.VERIFIED,
|
|
summary="The invariant is closed.",
|
|
security_invariant_closed=True,
|
|
review_basis="code_review",
|
|
regression_test_valid=True,
|
|
unit_test_coverage_valid=True,
|
|
reproduction_executed=False,
|
|
reproduction_summary="The vulnerable input is rejected.",
|
|
sibling_paths_reviewed=["app.py"],
|
|
preserved_behaviors=["The module compiles."],
|
|
regression_tests=[_regression()],
|
|
security_tests=[
|
|
CheckResult(
|
|
name="security reproduction",
|
|
argv=[sys.executable, "-c", "assert True"],
|
|
status=CheckStatus.PASSED,
|
|
exit_code=0,
|
|
duration_seconds=0,
|
|
target=VerificationTarget.PATCHED,
|
|
)
|
|
],
|
|
)
|
|
|
|
|
|
def test_candidate_from_legacy_report_preserves_draft_and_checks() -> None:
|
|
candidate = candidate_from_legacy_report(
|
|
{
|
|
"remediation_steps": "Reject unsafe input.",
|
|
"poc_description": "Call the vulnerable function.",
|
|
"fix_verification": "Reasoned about the patched branch.",
|
|
"code_locations": [
|
|
{
|
|
"file": "src/app.py",
|
|
"start_line": 9,
|
|
"end_line": 9,
|
|
"snippet": "sink(value)",
|
|
"fix_before": "sink(value)",
|
|
"fix_after": "sink(clean(value))",
|
|
}
|
|
],
|
|
}
|
|
)
|
|
|
|
assert candidate is not None
|
|
assert candidate.security_invariant == "Reject unsafe input."
|
|
assert candidate.draft_edits[0].after == "sink(clean(value))"
|
|
assert candidate.reported_checks[0].executed is False
|
|
assert candidate.known_gaps == ["The reporting-agent verification is not independent."]
|
|
|
|
|
|
def test_anchor_location_rewrites_invented_line_numbers(tmp_path: Path) -> None:
|
|
(tmp_path / "app.py").write_text("first\nsecond\ntarget\nlast\n", encoding="utf-8")
|
|
location = CandidateLocation(
|
|
file="app.py",
|
|
start_line=99,
|
|
end_line=99,
|
|
snippet="target",
|
|
)
|
|
|
|
result = anchor_location(tmp_path, location)
|
|
|
|
assert result.status is AnchorStatus.UNIQUE
|
|
assert result.location.start_line == 3
|
|
assert result.location.end_line == 3
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("content", "expected"),
|
|
[
|
|
("first\nlast\n", AnchorStatus.MISSING),
|
|
("target\nmiddle\ntarget\n", AnchorStatus.AMBIGUOUS),
|
|
],
|
|
)
|
|
def test_anchor_location_rejects_non_unique_source(
|
|
tmp_path: Path,
|
|
content: str,
|
|
expected: AnchorStatus,
|
|
) -> None:
|
|
(tmp_path / "app.py").write_text(content, encoding="utf-8")
|
|
edit = FixEdit(
|
|
file="app.py",
|
|
start_line=1,
|
|
end_line=1,
|
|
before="target",
|
|
after="safe",
|
|
)
|
|
|
|
assert anchor_location(tmp_path, edit).status is expected
|
|
|
|
|
|
def test_anchor_location_detects_stale_file_digest(tmp_path: Path) -> None:
|
|
original = "target\n"
|
|
(tmp_path / "app.py").write_text("changed\n", encoding="utf-8")
|
|
edit = FixEdit(
|
|
file="app.py",
|
|
start_line=1,
|
|
end_line=1,
|
|
before="target",
|
|
after="safe",
|
|
original_sha256=hashlib.sha256(original.encode()).hexdigest(),
|
|
)
|
|
|
|
assert anchor_location(tmp_path, edit).status is AnchorStatus.STALE
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_prepare_fix_rejects_wrong_source_commit(tmp_path: Path) -> None:
|
|
workspace, _commit = _workspace(tmp_path)
|
|
|
|
result = await prepare_fix(
|
|
_request(_candidate("0" * 40)),
|
|
workspace,
|
|
repair=_noop_repair,
|
|
verify=_verified,
|
|
)
|
|
|
|
assert result.state is PreparationState.STALE
|
|
assert result.blocker is not None
|
|
assert result.blocker.kind is BlockerKind.SOURCE
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_prepare_fix_rejects_dirty_workspace(tmp_path: Path) -> None:
|
|
workspace, commit = _workspace(tmp_path)
|
|
(workspace / "stray.txt").write_text("unrelated\n", encoding="utf-8")
|
|
|
|
result = await prepare_fix(
|
|
_request(_candidate(commit)),
|
|
workspace,
|
|
repair=_noop_repair,
|
|
verify=_verified,
|
|
)
|
|
|
|
assert result.state is PreparationState.STALE
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_build_git_manifest_lists_files_inside_new_directory(tmp_path: Path) -> None:
|
|
workspace, _commit = _workspace(tmp_path)
|
|
package = workspace / "pkg"
|
|
package.mkdir()
|
|
(package / "mod.py").write_text("x = 1\n", encoding="utf-8")
|
|
|
|
entries, _summary, _artifact = await build_git_manifest(workspace)
|
|
|
|
paths = {entry.path for entry in entries}
|
|
assert "pkg/mod.py" in paths
|
|
entry = next(entry for entry in entries if entry.path == "pkg/mod.py")
|
|
assert entry.operation == "add"
|
|
assert entry.resulting_sha256 is not None
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_run_command_drops_ambient_credentials(
|
|
tmp_path: Path,
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
monkeypatch.setenv("STRIX_AMBIENT_TOKEN", "hunter2")
|
|
command = CommandSpec(
|
|
name="env probe",
|
|
argv=[
|
|
sys.executable,
|
|
"-c",
|
|
"import os; print(os.environ.get('STRIX_AMBIENT_TOKEN', '<absent>'))",
|
|
],
|
|
)
|
|
|
|
sealed = await run_command(tmp_path, command, network_allowed=True)
|
|
assert sealed.status is CheckStatus.PASSED
|
|
assert "<absent>" in sealed.output
|
|
|
|
granted = await run_command(
|
|
tmp_path,
|
|
command,
|
|
credentials_allowed=["STRIX_AMBIENT_TOKEN"],
|
|
network_allowed=True,
|
|
)
|
|
assert granted.status is CheckStatus.PASSED
|
|
assert "hunter2" in granted.output
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_run_command_blocks_egress_when_network_not_allowed(tmp_path: Path) -> None:
|
|
command = CommandSpec(
|
|
name="egress probe",
|
|
argv=[
|
|
sys.executable,
|
|
"-c",
|
|
(
|
|
"import socket, sys; s = socket.socket(); s.settimeout(3); "
|
|
"sys.exit(0 if s.connect_ex(('1.1.1.1', 53)) == 0 else 1)"
|
|
),
|
|
],
|
|
)
|
|
|
|
result = await run_command(tmp_path, command)
|
|
|
|
if _network_isolation_prefix() is None:
|
|
assert result.status is CheckStatus.UNAVAILABLE
|
|
assert "not run" in result.output
|
|
else:
|
|
assert result.status is CheckStatus.FAILED
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_manifest_patch_includes_untracked_companion_file(tmp_path: Path) -> None:
|
|
workspace, _ = _workspace(tmp_path)
|
|
(workspace / "companion.py").write_text("guard = True\n")
|
|
manifest, _, _ = await build_git_manifest(workspace)
|
|
patch = await build_git_patch(workspace, manifest)
|
|
assert b"b/companion.py" in patch
|
|
assert b"+guard = True" in patch
|
|
_git(workspace, "diff", "--check")
|
|
|
|
|
|
def test_skipped_check_and_missing_runtime_are_not_passing_evidence() -> None:
|
|
assert command_status(0, "Skipping to avoid parser lock")[0] is CheckStatus.SKIPPED
|
|
assert command_status(127, "bun: command not found")[0] is CheckStatus.UNAVAILABLE
|
|
assert (
|
|
not _regression()
|
|
.model_copy(
|
|
update={"base": _regression().base.model_copy(update={"failure_kind": "environment"})}
|
|
)
|
|
.passed()
|
|
)
|
|
|
|
|
|
def test_candidate_keeps_full_finding_without_inventing_reproduction() -> None:
|
|
candidate = candidate_from_legacy_report(
|
|
{
|
|
"title": "Authorization bypass",
|
|
"technical_analysis": "Critical exploit context.",
|
|
"remediation_steps": "Enforce authorization.",
|
|
"code_locations": [
|
|
{
|
|
"file": "app.py",
|
|
"start_line": 1,
|
|
"end_line": 1,
|
|
"fix_before": "unsafe",
|
|
"fix_after": "safe",
|
|
}
|
|
],
|
|
}
|
|
)
|
|
assert candidate and candidate.finding
|
|
assert candidate.finding.description == "Critical exploit context."
|
|
assert candidate.reproduction is None
|
|
|
|
|
|
def test_ambiguous_imports_and_syntax_never_authorize_environment_recovery() -> None:
|
|
for output in [
|
|
"Cannot find module '/tmp/test/skills/policy.js'",
|
|
"No module named 'wrong_repo_path'",
|
|
"SyntaxError: invalid syntax",
|
|
]:
|
|
status, kind = command_status(1, output)
|
|
assert status is CheckStatus.FAILED
|
|
assert kind == "unknown"
|
|
assert command_status(127, "bun: command not found") == (CheckStatus.UNAVAILABLE, "environment")
|
|
|
|
|
|
def test_regression_requires_consistent_execution_provenance() -> None:
|
|
regression = _regression()
|
|
for leg in (regression.base, regression.patched, regression.behavior):
|
|
leg.source_digest = "a" * 64
|
|
leg.environment_id = "execution:0"
|
|
assert regression.passed()
|
|
regression.base.environment_id = "execution:1"
|
|
assert not regression.passed()
|
|
regression.base.environment_id = "execution:0"
|
|
regression.behavior = regression.behavior.model_copy(update={"source_digest": "b" * 64})
|
|
assert not regression.passed()
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_native_flow_runs_regression_before_review_and_saves_test(tmp_path: Path) -> None:
|
|
workspace, commit = _workspace(tmp_path)
|
|
|
|
async def review(context, checks):
|
|
assert checks[0].purpose == "regression"
|
|
assert checks[0].tests_passed == 1
|
|
assert checks[0].exit_code == 0
|
|
return await _verified(context, checks)
|
|
|
|
result = await prepare_fix(
|
|
_request(_candidate(commit)), workspace, repair=_noop_repair, verify=review
|
|
)
|
|
assert result.state is PreparationState.READY, result.model_dump_json()
|
|
assert result.validation_mode == "native_tests"
|
|
assert result.test_plan.regression_files == ["test_app.py"]
|
|
assert set(result.changed_files) == {"app.py", "test_app.py"}
|
|
assert result.verifier.review_basis == "code_review"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_failing_unit_suite_is_returned_to_repair_before_review(tmp_path: Path) -> None:
|
|
workspace, commit = _workspace(tmp_path)
|
|
calls = []
|
|
|
|
async def repair(context, checks):
|
|
calls.append(context.attempt)
|
|
outcome = await _noop_repair(context, checks)
|
|
(workspace / "test_existing.py").write_text(
|
|
"import unittest\nclass Existing(unittest.TestCase):\n"
|
|
f" def test_existing(self): self.assertTrue({context.attempt > 1})\n"
|
|
)
|
|
outcome.test_plan.unit_tests = [
|
|
CommandSpec(
|
|
name="existing suite", argv=[sys.executable, "-m", "unittest", "test_existing"]
|
|
)
|
|
]
|
|
outcome.test_plan.no_unit_tests_reason = None
|
|
if context.attempt == 2:
|
|
assert any(c.purpose == "unit" and c.status is CheckStatus.FAILED for c in checks)
|
|
return outcome
|
|
|
|
result = await prepare_fix(
|
|
_request(_candidate(commit)), workspace, repair=repair, verify=_verified
|
|
)
|
|
assert result.state is PreparationState.READY
|
|
assert calls == [1, 2]
|
|
assert result.attempt_history[0].verifier is None
|
|
assert result.checks[1].tests_passed == 1
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize(
|
|
"output", ["Ran 0 tests in 0.000s\n\nOK", "3 skipped", "collected 0 items"]
|
|
)
|
|
async def test_empty_or_skipped_tests_never_approve(tmp_path: Path, output: str) -> None:
|
|
workspace, commit = _workspace(tmp_path)
|
|
|
|
async def runner(_root, command):
|
|
return CheckResult(
|
|
name=command.name,
|
|
argv=command.argv,
|
|
status=CheckStatus.PASSED,
|
|
exit_code=0,
|
|
output=output,
|
|
duration_seconds=0,
|
|
)
|
|
|
|
async def review(*_args):
|
|
pytest.fail("Review must not run without passing tests")
|
|
|
|
result = await prepare_fix(
|
|
_request(_candidate(commit)),
|
|
workspace,
|
|
repair=_noop_repair,
|
|
verify=review,
|
|
command_runner=runner,
|
|
)
|
|
assert result.state is PreparationState.BLOCKED
|
|
assert result.final_file_manifest
|
|
assert result.checks[0].status is CheckStatus.SKIPPED
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_reviewer_concrete_defect_retries_and_reruns_tests(tmp_path: Path) -> None:
|
|
workspace, commit = _workspace(tmp_path)
|
|
|
|
async def review(context, checks):
|
|
if context.attempt == 1:
|
|
return VerifierResult(
|
|
decision=VerificationDecision.REJECTED,
|
|
summary="Missing sibling guard",
|
|
gaps=["Check the second entry point"],
|
|
repairable=True,
|
|
)
|
|
assert context.feedback[0].verifier.gaps == ["Check the second entry point"]
|
|
return await _verified(context, checks)
|
|
|
|
result = await prepare_fix(
|
|
_request(_candidate(commit)), workspace, repair=_noop_repair, verify=review
|
|
)
|
|
assert result.state is PreparationState.READY
|
|
assert len(result.attempt_history) == 2
|
|
assert all(a.checks[0].tests_passed == 1 for a in result.attempt_history)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_exhausted_repair_can_be_validated_from_saved_work(tmp_path: Path) -> None:
|
|
workspace, commit = _workspace(tmp_path)
|
|
|
|
async def repair(context, checks):
|
|
outcome = await _noop_repair(context, checks)
|
|
outcome.status = RepairStatus.BUDGET_EXHAUSTED
|
|
return outcome
|
|
|
|
result = await prepare_fix(
|
|
_request(_candidate(commit)), workspace, repair=repair, verify=_verified
|
|
)
|
|
assert result.state is PreparationState.READY
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize(
|
|
"defect",
|
|
[
|
|
"missing_plan",
|
|
"missing_file",
|
|
"rejected_test",
|
|
"inconclusive_review",
|
|
"exception",
|
|
"source_change",
|
|
],
|
|
)
|
|
async def test_incomplete_work_is_retained_without_ready(tmp_path: Path, defect: str) -> None:
|
|
workspace, commit = _workspace(tmp_path)
|
|
|
|
async def repair(context, checks):
|
|
outcome = await _noop_repair(context, checks)
|
|
if defect == "missing_plan":
|
|
outcome.test_plan = None
|
|
if defect == "missing_file":
|
|
outcome.test_plan.regression_files = ["invented.py"]
|
|
return outcome
|
|
|
|
async def review(context, checks):
|
|
if defect == "exception":
|
|
raise RuntimeError("review provider failed")
|
|
if defect == "source_change":
|
|
(workspace / "app.py").write_text("bad=True\n")
|
|
result = await _verified(context, checks)
|
|
if defect == "rejected_test":
|
|
result.regression_test_valid = False
|
|
if defect == "inconclusive_review":
|
|
result.decision = VerificationDecision.INCONCLUSIVE
|
|
return result
|
|
|
|
result = await prepare_fix(
|
|
_request(_candidate(commit)), workspace, repair=repair, verify=review
|
|
)
|
|
assert result.state in {PreparationState.BLOCKED, PreparationState.FAILED}
|
|
assert result.final_file_manifest
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("output", "expected"),
|
|
[
|
|
("Tests: 2 passed, 2 total", 2),
|
|
("2 pass\n0 fail", 2),
|
|
("# tests 2\n# pass 2\n# fail 0", 2),
|
|
("Ran 3 tests in 0.1s\n\nOK (skipped=1)", 2),
|
|
("3 examples, 0 failures, 1 pending", 2),
|
|
("--- PASS: TestSafe (0.00s)", 1),
|
|
("Ran 2 tests in 0.1s\n\nOK (skipped=2)", 0),
|
|
("No tests found", None),
|
|
],
|
|
)
|
|
def test_runner_summaries_record_actual_execution(output: str, expected: int | None) -> None:
|
|
|
|
assert passed_test_count(output) == expected
|
|
|
|
|
|
def test_new_command_metadata_does_not_change_existing_finding_digest(tmp_path: Path) -> None:
|
|
|
|
_root, commit = _workspace(tmp_path)
|
|
candidate = _candidate(commit)
|
|
payload = candidate.model_dump(mode="json")
|
|
payload.pop("finding")
|
|
payload["reproduction"]["command"].pop("purpose")
|
|
previous = hashlib.sha256(
|
|
json.dumps(payload, sort_keys=True, separators=(",", ":")).encode()
|
|
).hexdigest()
|
|
assert candidate.digest() == previous
|
|
|
|
|
|
def test_unknown_runner_format_preserves_exit_result_for_independent_review() -> None:
|
|
|
|
command = CommandSpec(name="custom runner", argv=["./tests/run"], purpose="unit")
|
|
result = record_test_execution(
|
|
CheckResult(
|
|
name=command.name,
|
|
argv=command.argv,
|
|
status=CheckStatus.PASSED,
|
|
exit_code=0,
|
|
duration_seconds=1,
|
|
output="All project assertions completed successfully.",
|
|
),
|
|
command,
|
|
)
|
|
assert result.status is CheckStatus.PASSED
|
|
assert result.tests_passed is None
|