Remove retired fix execution and proof machinery

This commit is contained in:
Jonathan Singer 2026-09-29 23:26:23 -04:00
parent 348fbf2608
commit 5badb2d541
5 changed files with 43 additions and 573 deletions

View file

@ -15,24 +15,18 @@ from strix.fix.contracts import (
FixPreparationResultV1,
PreparationBlocker,
PreparationState,
RegressionTestResult,
RepairOutcome,
RepairStatus,
RepositoryTestPlan,
ReproductionSpec,
ReviewConcern,
SourceIdentity,
SourceIdentityKind,
VerificationDecision,
VerificationHarness,
VerificationTarget,
VerifierResult,
candidate_from_legacy_report,
)
from strix.fix.prepare import (
PreparationCancelledError,
PreparationContext,
PreparationPolicy,
build_git_manifest,
build_git_patch,
prepare_fix,
@ -56,19 +50,13 @@ __all__ = [
"PreparationBlocker",
"PreparationCancelledError",
"PreparationContext",
"PreparationPolicy",
"PreparationState",
"RegressionTestResult",
"RepairOutcome",
"RepairStatus",
"RepositoryTestPlan",
"ReproductionSpec",
"ReviewConcern",
"SourceIdentity",
"SourceIdentityKind",
"VerificationDecision",
"VerificationHarness",
"VerificationTarget",
"VerifierResult",
"build_git_manifest",
"build_git_patch",

View file

@ -42,11 +42,6 @@ class CheckStatus(StrEnum):
SKIPPED = "skipped"
class VerificationTarget(StrEnum):
BASE = "base"
PATCHED = "patched"
class VerificationDecision(StrEnum):
VERIFIED = "verified"
REJECTED = "rejected"
@ -152,28 +147,6 @@ class ReproductionSpec(ContractModel):
command: CommandSpec | None = None
class RepositoryTestPlan(ContractModel):
"""Historical native-test handoff; agent-driven runs record commands directly."""
regression_test: CommandSpec
regression_files: list[str] = Field(min_length=1)
unit_tests: list[CommandSpec] = []
no_unit_tests_reason: str | None = None
@field_validator("regression_files")
@classmethod
def validate_files(cls, paths: list[str]) -> list[str]:
return list(dict.fromkeys(_validate_relative_path(path) for path in paths))
@model_validator(mode="after")
def explain_missing_suite(self) -> RepositoryTestPlan:
if not self.unit_tests and not (self.no_unit_tests_reason or "").strip():
raise ValueError(
"Provide the existing unit-test commands or explain why no suite exists."
)
return self
class ReportedCheck(ContractModel):
name: str
result: str
@ -230,6 +203,8 @@ class FixPreparationRequestV1(ContractModel):
class CheckResult(ContractModel):
"""Recorded native shell execution; the reviewer interprets its meaning."""
name: str
argv: list[str]
status: CheckStatus
@ -237,90 +212,17 @@ class CheckResult(ContractModel):
duration_seconds: float = Field(ge=0)
output: str = ""
required: bool = True
target: VerificationTarget | None = None
baseline_status: CheckStatus | None = None
baseline_output: str | None = None
cwd: str = "."
baseline_exit_code: int | None = None
failure_kind: (
Literal["environment", "timeout", "check", "source_changed", "harness", "unknown"] | None
) = None
workspace_root: str | None = None
baseline_workspace_root: str | None = None
source_digest: str | None = Field(default=None, pattern=r"^[0-9a-f]{64}$")
environment_id: str | None = None
purpose: Literal["quality", "regression", "unit", "security"] = "quality"
tests_passed: int | None = Field(default=None, ge=0)
class RegressionTestResult(ContractModel):
"""One unchanged test run on both revisions, plus a legitimate-operation check."""
name: str
expected_base_failure: str = Field(min_length=1)
harness_sha256: str = Field(pattern=r"^[0-9a-f]{64}$")
base: CheckResult
patched: CheckResult
behavior: CheckResult
def passed(self) -> bool:
results = (self.base, self.patched, self.behavior)
# Legacy records remain readable. Once provenance is supplied, every leg
# must have it and the environment must be unchanged across the pair.
if any(item.source_digest or item.environment_id for item in results) and (
not all(item.source_digest and item.environment_id for item in results)
or len({item.environment_id for item in results}) != 1
or self.patched.source_digest != self.behavior.source_digest
):
return False
return (
self.base.target is VerificationTarget.BASE
and self.patched.target is VerificationTarget.PATCHED
and self.behavior.target is VerificationTarget.PATCHED
and self.base.argv == self.patched.argv
and self.base.cwd == self.patched.cwd
and self.base.status is CheckStatus.FAILED
and self.base.exit_code == 1
and self.base.failure_kind in {None, "check"}
and self.expected_base_failure in self.base.output
and self.patched.status is CheckStatus.PASSED
and self.patched.exit_code == 0
and self.behavior.status is CheckStatus.PASSED
and self.behavior.exit_code == 0
)
class VerificationHarness(ContractModel):
path: str
sha256: str = Field(pattern=r"^[0-9a-f]{64}$")
content: str
class ReviewConcern(ContractModel):
kind: Literal["repair_needed", "customer_prerequisite", "optional_follow_up"]
summary: str = Field(min_length=1)
class VerifierResult(ContractModel):
decision: VerificationDecision
summary: str
security_invariant_closed: bool = False
reproduction_executed: bool = False
reproduction_summary: str | None = None
sibling_paths_reviewed: list[str] = []
preserved_behaviors: list[str] = []
gaps: list[str] = []
security_tests: list[CheckResult] = []
repairable: bool = False
blocker: PreparationBlocker | None = None
regression_tests: list[RegressionTestResult] = []
harnesses: list[VerificationHarness] = []
notes: list[str] = []
review_basis: Literal["execution", "code_review"] | None = None
regression_test_valid: bool = False
unit_test_coverage_valid: bool = False
# None preserves historical reviews; new reviewers classify every concern.
concerns: list[ReviewConcern] | None = None
source_digest: str | None = Field(default=None, pattern=r"^[0-9a-f]{64}$")
turns_used: int = Field(default=0, ge=0)
@ -329,12 +231,9 @@ class RepairOutcome(ContractModel):
status: RepairStatus
summary: str = Field(min_length=1)
gaps: list[str] = []
reproduction_command: CommandSpec | None = None
turns_used: int = Field(default=0, ge=0)
blocker: PreparationBlocker | None = None
checks: list[CommandSpec] = []
command_results: list[CheckResult] = []
test_plan: RepositoryTestPlan | None = None
source_digest: str | None = Field(default=None, pattern=r"^[0-9a-f]{64}$")
@ -342,7 +241,6 @@ class FixPreparationAttempt(ContractModel):
attempt: int = Field(ge=1)
repair: RepairOutcome
checks: list[CheckResult] = []
security_reproduction: CheckResult | None = None
verifier: VerifierResult | None = None
workspace_digest: str = Field(pattern=r"^[0-9a-f]{64}$")
@ -359,8 +257,7 @@ class FileManifestEntry(ContractModel):
class FixPreparationResultV1(ContractModel):
version: Literal["1"] = "1"
# Absent on historical records; keep their stronger, paired-proof interpretation.
validation_mode: Literal["paired", "native_tests", "agent_review"] = "paired"
validation_mode: Literal["agent_review"] = "agent_review"
prepared_source_digest: str | None = Field(default=None, pattern=r"^[0-9a-f]{64}$")
state: PreparationState
stop_reason: str
@ -372,16 +269,13 @@ class FixPreparationResultV1(ContractModel):
changed_files: list[str] = []
diff_summary: str = ""
checks: list[CheckResult] = []
security_reproduction: CheckResult | None = None
verifier: VerifierResult | None = None
attempt_history: list[FixPreparationAttempt] = []
gaps: list[str] = []
blocker: PreparationBlocker | None = None
setup_checks: list[CheckResult] = []
attempts: int = Field(default=0, ge=0)
elapsed_seconds: float = Field(default=0, ge=0)
cost_usd: float | None = Field(default=None, ge=0)
test_plan: RepositoryTestPlan | None = None
def candidate_from_legacy_report(

View file

@ -1,93 +0,0 @@
"""Execution facts shared by local and managed fix preparation."""
from __future__ import annotations
import re
from typing import Literal
from strix.fix.contracts import CheckResult, CheckStatus, CommandSpec
def passed_test_count(output: str) -> int | None: # noqa: PLR0911 - native summary formats
"""Recognize native test-runner summaries, not the model's account of a run.
Counts are optional: runners use different output formats. This is execution
evidence, not an assertion that a test covers the finding; the reviewer checks that.
"""
plain = re.sub(r"\x1b\[[0-9;]*[a-zA-Z]", "", output)
if command_status(0, plain)[0] is CheckStatus.SKIPPED:
return 0
counts = re.findall(r"\b(\d+)\s+(?:passed|pass)\b", plain, re.IGNORECASE)
counts += re.findall(r"(?:#|\u2139)\s+pass\s+(\d+)\b", plain)
if counts:
return max(int(count) for count in counts)
unittest = re.search(r"Ran (\d+) tests? in [^\n]+\n\s*OK(?: \(skipped=(\d+)\))?", plain)
if unittest:
return max(0, int(unittest[1]) - int(unittest[2] or 0))
rspec = re.search(r"(\d+) examples?, (\d+) failures?(?:, (\d+) pending)?", plain)
if rspec:
return max(0, int(rspec[1]) - int(rspec[2]) - int(rspec[3] or 0))
minitest = re.search(
r"(\d+) runs, \d+ assertions, (\d+) failures, (\d+) errors, (\d+) skips", plain
)
if minitest:
return max(0, int(minitest[1]) - sum(int(minitest[i]) for i in (2, 3, 4)))
if re.search(r"^(?:=+\s*)?[1-9]\d* skipped(?:\s+in [^\n=]+)?(?:\s*=+)?$", plain, re.MULTILINE):
return 0
go = re.findall(r"^\s*--- PASS: ", plain, re.MULTILINE)
return len(go) if go else None
def record_test_execution(result: CheckResult, command: CommandSpec) -> CheckResult:
"""Known empty/skipped runs cannot pass; unknown summary formats go to review."""
count = passed_test_count(result.output) if command.purpose != "quality" else None
update: dict[str, object] = {
"purpose": command.purpose,
"tests_passed": count,
"required": command.required,
"name": command.name,
"argv": command.argv,
"cwd": command.cwd,
}
if command.purpose != "quality" and result.status is CheckStatus.PASSED and count == 0:
update.update(
status=CheckStatus.SKIPPED,
output=result.output + "\nThe runner reported no passing tests.",
)
return result.model_copy(update=update)
def command_status(
exit_code: int, output: str
) -> tuple[CheckStatus, Literal["environment", "check", "harness", "unknown"] | None]:
"""Classify launch/collection failures before interpreting an assertion result."""
if exit_code in {126, 127} or (
exit_code != 0
and re.search(
r"(?:command not found|exec: \S+: not found)",
output,
re.IGNORECASE,
)
):
return CheckStatus.UNAVAILABLE, "environment"
if exit_code != 0 and re.search(
r"(?:SyntaxError|ReferenceError: require is not defined)", output
):
return CheckStatus.FAILED, "unknown"
if exit_code != 0 and re.search(
r"(?:No module named |Cannot find module |ERR_MODULE_NOT_FOUND|"
r"ECONNREFUSED|Could not connect to server)",
output,
re.IGNORECASE,
):
# An incorrect import or a failed service is not evidence that installing
# dependencies is appropriate. Return it to the caller for diagnosis.
return CheckStatus.FAILED, "unknown"
if re.search(
r"(?:Skipping to avoid parser lock|collected 0 items|"
r"No test files found|No tests found, exiting with code 0)",
output,
re.IGNORECASE,
):
return CheckStatus.SKIPPED, None
return (CheckStatus.PASSED, None) if exit_code == 0 else (CheckStatus.FAILED, "check")

View file

@ -3,14 +3,12 @@
from __future__ import annotations
import asyncio
import functools
import hashlib
import json
import logging
import os
import subprocess
import time
from collections.abc import Awaitable, Callable, Iterable
from collections.abc import Awaitable, Callable
from dataclasses import dataclass, field
from pathlib import Path
from typing import Literal
@ -18,8 +16,6 @@ from typing import Literal
from strix.fix.contracts import (
BlockerKind,
CheckResult,
CheckStatus,
CommandSpec,
FileManifestEntry,
FixCandidateV1,
FixPreparationAttempt,
@ -32,24 +28,16 @@ from strix.fix.contracts import (
VerificationDecision,
VerifierResult,
)
from strix.fix.evidence import command_status
class PreparationCancelledError(RuntimeError):
pass
CommandRunner = Callable[[Path, CommandSpec], Awaitable[CheckResult]]
ManifestBuilder = Callable[[Path], Awaitable[tuple[list[FileManifestEntry], str, str | None]]]
CancellationCheck = Callable[[], bool]
@dataclass(slots=True)
class PreparationPolicy:
timeout_seconds: int = 7200
max_output_chars: int = 20000
@dataclass(slots=True)
class PreparationContext:
request: FixPreparationRequestV1
@ -61,7 +49,7 @@ class PreparationContext:
RepairAgent = Callable[
[PreparationContext, list[CheckResult]],
Awaitable[RepairOutcome | None],
Awaitable[RepairOutcome],
]
IndependentVerifier = Callable[
[PreparationContext, list[CheckResult]],
@ -71,127 +59,6 @@ SourceVerifier = Callable[[PreparationContext], Awaitable[bool]]
EvidenceReader = Callable[[], Awaitable[list[CheckResult]]]
_COMMAND_ENV_ALLOWLIST = frozenset(
{
"HOME",
"LANG",
"LC_ALL",
"PATH",
"PYTHONHOME",
"PYTHONPATH",
"SYSTEMROOT",
"TEMP",
"TMP",
"TMPDIR",
"VIRTUAL_ENV",
"SystemRoot",
}
)
@functools.lru_cache(maxsize=1)
def _network_isolation_prefix() -> tuple[str, ...] | None:
"""Return a working ``unshare`` prefix that creates an empty network
namespace, or None when the platform cannot isolate egress."""
for prefix in (("unshare", "-Urn"), ("unshare", "-n")):
try:
probe = subprocess.run( # noqa: S603
[*prefix, "true"],
capture_output=True,
timeout=15,
check=False,
)
except (OSError, subprocess.SubprocessError):
continue
if probe.returncode == 0:
return prefix
return None
def _command_environment(credentials_allowed: Iterable[str]) -> dict[str, str]:
allowed = _COMMAND_ENV_ALLOWLIST | set(credentials_allowed)
env = {key: value for key, value in os.environ.items() if key in allowed}
env["PYTHONDONTWRITEBYTECODE"] = "1"
return env
async def run_command(
workspace: Path,
command: CommandSpec,
*,
credentials_allowed: Iterable[str] = (),
network_allowed: bool = False,
) -> CheckResult:
started = time.monotonic()
cwd = (workspace / command.cwd).resolve()
if not cwd.is_relative_to(workspace.resolve()) or not cwd.is_dir():
return CheckResult(
name=command.name,
argv=command.argv,
status=CheckStatus.UNAVAILABLE,
duration_seconds=time.monotonic() - started,
output="The command working directory is unavailable.",
required=command.required,
)
argv = list(command.argv)
if not network_allowed:
prefix = _network_isolation_prefix()
if prefix is None:
return CheckResult(
name=command.name,
argv=command.argv,
status=CheckStatus.UNAVAILABLE,
duration_seconds=time.monotonic() - started,
output="Network isolation is unavailable, so the command was not run.",
required=command.required,
)
argv = [*prefix, *argv]
process: asyncio.subprocess.Process | None = None
try:
process = await asyncio.create_subprocess_exec(
*argv,
cwd=cwd,
env=_command_environment(credentials_allowed),
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.STDOUT,
)
output, _ = await asyncio.wait_for(process.communicate(), command.timeout_seconds)
except (FileNotFoundError, PermissionError) as exc:
return CheckResult(
name=command.name,
argv=command.argv,
status=CheckStatus.UNAVAILABLE,
duration_seconds=time.monotonic() - started,
output=str(exc),
required=command.required,
)
except TimeoutError:
if process is not None:
process.kill()
await process.wait()
return CheckResult(
name=command.name,
argv=command.argv,
status=CheckStatus.FAILED,
duration_seconds=time.monotonic() - started,
output=f"Timed out after {command.timeout_seconds} seconds.",
required=command.required,
)
status, failure_kind = command_status(process.returncode or 0, output.decode(errors="replace"))
return CheckResult(
name=command.name,
argv=command.argv,
status=status,
exit_code=process.returncode,
duration_seconds=time.monotonic() - started,
output=output.decode(errors="replace")[-20000:],
required=command.required,
failure_kind=failure_kind,
cwd=command.cwd,
workspace_root=str(workspace.resolve()),
)
async def build_git_manifest(
workspace: Path,
) -> tuple[list[FileManifestEntry], str, str | None]:
@ -361,7 +228,6 @@ def _result(
reason: str,
started: float,
checks: list[CheckResult] | None = None,
reproduction: CheckResult | None = None,
verifier: VerifierResult | None = None,
gaps: list[str] | None = None,
manifest: list[FileManifestEntry] | None = None,
@ -380,7 +246,6 @@ def _result(
if context.feedback
else None
),
test_plan=context.feedback[-1].repair.test_plan if context.feedback else None,
stop_reason=reason,
source_identity=context.candidate.source_identity,
candidate=context.candidate,
@ -390,7 +255,6 @@ def _result(
changed_files=[entry.path for entry in manifest or []],
diff_summary=diff_summary,
checks=checks or [],
security_reproduction=reproduction,
verifier=verifier,
attempt_history=attempt_history or [],
gaps=gaps or [],
@ -400,15 +264,6 @@ def _result(
)
def _repair_outcome(value: RepairOutcome | None) -> RepairOutcome:
if value is not None:
return value
return RepairOutcome(
status=RepairStatus.COMPLETE,
summary="The repair implementation returned control for independent evaluation.",
)
async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup
request: FixPreparationRequestV1,
workspace: Path,
@ -419,11 +274,9 @@ async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup
source_verifier: SourceVerifier = _verify_source,
evidence_reader: EvidenceReader | None = None,
cancelled: CancellationCheck = lambda: False,
policy: PreparationPolicy | None = None,
) -> FixPreparationResultV1:
"""Run repair/review conversations; agents own setup, tests and corrections."""
started = time.monotonic()
resolved_policy = policy or PreparationPolicy(timeout_seconds=request.timeout_seconds)
context = PreparationContext(request=request, workspace=workspace, candidate=request.candidate)
checks: list[CheckResult] = []
verifier: VerifierResult | None = None
@ -453,7 +306,6 @@ async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup
attempt_history=context.feedback,
blocker=blocker,
gaps=gaps,
reproduction=next((c for c in checks if c.purpose == "regression"), None),
)
async def execute() -> FixPreparationResultV1: # noqa: PLR0911 - explicit terminal outcomes
@ -482,7 +334,7 @@ async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup
workspace_digest=await workspace_digest(workspace),
)
context.feedback.append(record)
record.repair = _repair_outcome(await repair(context, checks))
record.repair = await repair(context, checks)
repair_turns += max(1, record.repair.turns_used)
checks = await evidence_reader() if evidence_reader else record.repair.command_results
record.checks = list(checks)
@ -539,7 +391,7 @@ async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup
)
try:
async with asyncio.timeout(resolved_policy.timeout_seconds):
async with asyncio.timeout(request.timeout_seconds):
return await execute()
except PreparationCancelledError:
return await finish(PreparationState.FAILED, "Fix preparation was cancelled.")

View file

@ -21,27 +21,21 @@ from strix.fix.contracts import (
FixEdit,
FixPreparationRequestV1,
PreparationState,
RegressionTestResult,
RepairOutcome,
RepairStatus,
ReproductionSpec,
SourceIdentity,
SourceIdentityKind,
VerificationDecision,
VerificationTarget,
VerifierResult,
candidate_from_legacy_report,
)
from strix.fix.evidence import command_status, passed_test_count, record_test_execution
from strix.fix.locations import AnchorStatus, anchor_location
from strix.fix.prepare import (
PreparationContext,
PreparationPolicy,
_network_isolation_prefix,
build_git_manifest,
build_git_patch,
prepare_fix,
run_command,
workspace_digest,
)
@ -50,34 +44,6 @@ if TYPE_CHECKING:
from pathlib import Path
def _regression() -> RegressionTestResult:
base = CheckResult(
name="authorization",
argv=["python", "regression.py"],
status=CheckStatus.FAILED,
exit_code=1,
duration_seconds=0,
target=VerificationTarget.BASE,
output="unauthorized access allowed",
)
passed = base.model_copy(
update={
"status": CheckStatus.PASSED,
"exit_code": 0,
"target": VerificationTarget.PATCHED,
"output": "passed",
}
)
return RegressionTestResult(
name="authorization",
expected_base_failure="unauthorized access allowed",
harness_sha256="a" * 64,
base=base,
patched=passed,
behavior=passed,
)
def _git(workspace: Path, *args: str) -> str:
return subprocess.run( # noqa: S603
["/usr/bin/git", *args],
@ -194,14 +160,7 @@ async def _noop_repair(
*context.request.checks,
]
digest = await workspace_digest(context.workspace)
results = []
for command in commands:
result = record_test_execution(
await run_command(context.workspace, command, network_allowed=True), command
)
results.append(
result.model_copy(update={"source_digest": digest, "environment_id": "test"})
)
results = [await _fixture_command(context.workspace, command) for command in commands]
return RepairOutcome(
status=RepairStatus.COMPLETE,
summary="Fixed and tested.",
@ -211,33 +170,36 @@ async def _noop_repair(
)
async def _fixture_command(workspace: Path, command: CommandSpec) -> CheckResult:
"""Run this module's synthetic fixture tests, without a production command wrapper."""
process = await asyncio.create_subprocess_exec(
command.argv[0],
"-B",
*command.argv[1:],
cwd=workspace,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.STDOUT,
)
output, _ = await process.communicate()
return CheckResult(
name=command.name,
argv=command.argv,
status=CheckStatus.PASSED if process.returncode == 0 else CheckStatus.FAILED,
exit_code=process.returncode,
duration_seconds=0,
output=output.decode(),
)
async def _verified(
_context: PreparationContext,
context: PreparationContext,
_checks: list[CheckResult],
) -> VerifierResult:
return VerifierResult(
decision=VerificationDecision.VERIFIED,
source_digest=_context.feedback[-1].repair.source_digest,
summary="The invariant is closed.",
security_invariant_closed=True,
source_digest=context.feedback[-1].repair.source_digest,
summary="The fix addresses the finding.",
review_basis="code_review",
regression_test_valid=True,
unit_test_coverage_valid=True,
reproduction_executed=False,
reproduction_summary="The vulnerable input is rejected.",
sibling_paths_reviewed=["app.py"],
preserved_behaviors=["The module compiles."],
regression_tests=[_regression()],
security_tests=[
CheckResult(
name="security reproduction",
argv=[sys.executable, "-c", "assert True"],
status=CheckStatus.PASSED,
exit_code=0,
duration_seconds=0,
target=VerificationTarget.PATCHED,
)
],
)
@ -369,58 +331,6 @@ async def test_build_git_manifest_lists_files_inside_new_directory(tmp_path: Pat
assert entry.resulting_sha256 is not None
@pytest.mark.asyncio
async def test_run_command_drops_ambient_credentials(
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setenv("STRIX_AMBIENT_TOKEN", "hunter2")
command = CommandSpec(
name="env probe",
argv=[
sys.executable,
"-c",
"import os; print(os.environ.get('STRIX_AMBIENT_TOKEN', '<absent>'))",
],
)
sealed = await run_command(tmp_path, command, network_allowed=True)
assert sealed.status is CheckStatus.PASSED
assert "<absent>" in sealed.output
granted = await run_command(
tmp_path,
command,
credentials_allowed=["STRIX_AMBIENT_TOKEN"],
network_allowed=True,
)
assert granted.status is CheckStatus.PASSED
assert "hunter2" in granted.output
@pytest.mark.asyncio
async def test_run_command_blocks_egress_when_network_not_allowed(tmp_path: Path) -> None:
command = CommandSpec(
name="egress probe",
argv=[
sys.executable,
"-c",
(
"import socket, sys; s = socket.socket(); s.settimeout(3); "
"sys.exit(0 if s.connect_ex(('1.1.1.1', 53)) == 0 else 1)"
),
],
)
result = await run_command(tmp_path, command)
if _network_isolation_prefix() is None:
assert result.status is CheckStatus.UNAVAILABLE
assert "not run" in result.output
else:
assert result.status is CheckStatus.FAILED
@pytest.mark.asyncio
async def test_manifest_patch_includes_untracked_companion_file(tmp_path: Path) -> None:
workspace, _ = _workspace(tmp_path)
@ -432,18 +342,6 @@ async def test_manifest_patch_includes_untracked_companion_file(tmp_path: Path)
_git(workspace, "diff", "--check")
def test_skipped_check_and_missing_runtime_are_not_passing_evidence() -> None:
assert command_status(0, "Skipping to avoid parser lock")[0] is CheckStatus.SKIPPED
assert command_status(127, "bun: command not found")[0] is CheckStatus.UNAVAILABLE
assert (
not _regression()
.model_copy(
update={"base": _regression().base.model_copy(update={"failure_kind": "environment"})}
)
.passed()
)
def test_candidate_keeps_full_finding_without_inventing_reproduction() -> None:
candidate = candidate_from_legacy_report(
{
@ -466,31 +364,6 @@ def test_candidate_keeps_full_finding_without_inventing_reproduction() -> None:
assert candidate.reproduction is None
def test_ambiguous_imports_and_syntax_never_authorize_environment_recovery() -> None:
for output in [
"Cannot find module '/tmp/test/skills/policy.js'",
"No module named 'wrong_repo_path'",
"SyntaxError: invalid syntax",
]:
status, kind = command_status(1, output)
assert status is CheckStatus.FAILED
assert kind == "unknown"
assert command_status(127, "bun: command not found") == (CheckStatus.UNAVAILABLE, "environment")
def test_regression_requires_consistent_execution_provenance() -> None:
regression = _regression()
for leg in (regression.base, regression.patched, regression.behavior):
leg.source_digest = "a" * 64
leg.environment_id = "execution:0"
assert regression.passed()
regression.base.environment_id = "execution:1"
assert not regression.passed()
regression.base.environment_id = "execution:0"
regression.behavior = regression.behavior.model_copy(update={"source_digest": "b" * 64})
assert not regression.passed()
@pytest.mark.asyncio
async def test_agent_tests_are_reused_without_controller_execution(tmp_path: Path) -> None:
workspace, commit = _workspace(tmp_path)
@ -506,9 +379,8 @@ async def test_agent_tests_are_reused_without_controller_execution(tmp_path: Pat
)
assert result.state is PreparationState.READY
assert result.validation_mode == "agent_review"
assert result.test_plan is None
assert result.checks == outcome.command_results
assert result.checks[0].tests_passed == 1
assert "Ran 1 test" in result.checks[0].output
assert result.prepared_source_digest == result.verifier.source_digest
@ -534,7 +406,7 @@ async def test_review_can_request_more_than_two_repairs(tmp_path: Path) -> None:
@pytest.mark.asyncio
@pytest.mark.parametrize("defect", ["missing_unit", "failed", "empty", "stale", "missing_request"])
@pytest.mark.parametrize("defect", ["missing_unit", "failed", "missing_request"])
async def test_agent_approval_owns_test_evidence_without_controller_retries(
tmp_path: Path, defect: str
) -> None:
@ -543,17 +415,14 @@ async def test_agent_approval_owns_test_evidence_without_controller_retries(
async def repair(context, checks):
result = await _noop_repair(context, checks)
if defect == "missing_unit":
result.command_results = [c for c in result.command_results if c.purpose != "unit"]
result.command_results = [
c for c in result.command_results if c.name != "existing suite"
]
elif defect == "missing_request":
result.command_results = [c for c in result.command_results if c.purpose != "quality"]
result.command_results = [c for c in result.command_results if c.name != "compile"]
else:
check = result.command_results[0]
if defect == "failed":
check.status, check.exit_code = CheckStatus.FAILED, 1
elif defect == "empty":
check.tests_passed = 0
else:
check.source_digest = "a" * 64
check.status, check.exit_code = CheckStatus.FAILED, 1
return result
request = _request(_candidate(commit))
@ -581,12 +450,7 @@ async def test_incomplete_validation_can_be_finished_by_reviewer(tmp_path: Path)
async def review(context, checks):
assert checks[1].status is CheckStatus.FAILED
command = CommandSpec(name="existing suite", argv=checks[1].argv, purpose="unit")
executed = record_test_execution(
await run_command(workspace, command, network_allowed=True), command
)
evidence[1] = executed.model_copy(
update={"source_digest": evidence[0].source_digest, "environment_id": "test"}
)
evidence[1] = await _fixture_command(workspace, command)
return await _verified(context, checks)
async def read_evidence():
@ -631,12 +495,13 @@ async def test_interruptions_preserve_partial_patch_without_approval(
raise AssertionError("Stopped repair must not be approved")
result = await prepare_fix(
_request(_candidate(commit)),
_request(_candidate(commit)).model_copy(
update={"timeout_seconds": 1 if stop == "timeout" else 30}
),
workspace,
repair=repair,
verify=review,
cancelled=lambda: cancel,
policy=PreparationPolicy(timeout_seconds=1 if stop == "timeout" else 30),
)
assert result.state in {PreparationState.BLOCKED, PreparationState.FAILED}
assert result.final_file_manifest
@ -686,24 +551,6 @@ async def test_empty_deliverable_does_not_start_another_repair(tmp_path: Path) -
assert "without a deliverable patch" in result.stop_reason
@pytest.mark.parametrize(
("output", "expected"),
[
("Tests: 2 passed, 2 total", 2),
("2 pass\n0 fail", 2),
("# tests 2\n# pass 2\n# fail 0", 2),
("Ran 3 tests in 0.1s\n\nOK (skipped=1)", 2),
("3 examples, 0 failures, 1 pending", 2),
("--- PASS: TestSafe (0.00s)", 1),
("Ran 2 tests in 0.1s\n\nOK (skipped=2)", 0),
("No tests found", None),
],
)
def test_runner_summaries_record_actual_execution(output: str, expected: int | None) -> None:
assert passed_test_count(output) == expected
def test_new_command_metadata_does_not_change_existing_finding_digest(tmp_path: Path) -> None:
_root, commit = _workspace(tmp_path)
@ -715,21 +562,3 @@ def test_new_command_metadata_does_not_change_existing_finding_digest(tmp_path:
json.dumps(payload, sort_keys=True, separators=(",", ":")).encode()
).hexdigest()
assert candidate.digest() == previous
def test_unknown_runner_format_preserves_exit_result_for_independent_review() -> None:
command = CommandSpec(name="custom runner", argv=["./tests/run"], purpose="unit")
result = record_test_execution(
CheckResult(
name=command.name,
argv=command.argv,
status=CheckStatus.PASSED,
exit_code=0,
duration_seconds=1,
output="All project assertions completed successfully.",
),
command,
)
assert result.status is CheckStatus.PASSED
assert result.tests_passed is None