From 5badb2d5414bb18717196d9744d17a52a3877fa9 Mon Sep 17 00:00:00 2001 From: Jonathan Singer Date: Tue, 29 Sep 2026 23:26:23 -0400 Subject: [PATCH] Remove retired fix execution and proof machinery --- strix/fix/__init__.py | 12 -- strix/fix/contracts.py | 112 +--------------- strix/fix/evidence.py | 93 ------------- strix/fix/prepare.py | 156 +--------------------- tests/test_fix_preparation.py | 243 +++++----------------------------- 5 files changed, 43 insertions(+), 573 deletions(-) delete mode 100644 strix/fix/evidence.py diff --git a/strix/fix/__init__.py b/strix/fix/__init__.py index fcd99490d..5b77605a2 100644 --- a/strix/fix/__init__.py +++ b/strix/fix/__init__.py @@ -15,24 +15,18 @@ from strix.fix.contracts import ( FixPreparationResultV1, PreparationBlocker, PreparationState, - RegressionTestResult, RepairOutcome, RepairStatus, - RepositoryTestPlan, ReproductionSpec, - ReviewConcern, SourceIdentity, SourceIdentityKind, VerificationDecision, - VerificationHarness, - VerificationTarget, VerifierResult, candidate_from_legacy_report, ) from strix.fix.prepare import ( PreparationCancelledError, PreparationContext, - PreparationPolicy, build_git_manifest, build_git_patch, prepare_fix, @@ -56,19 +50,13 @@ __all__ = [ "PreparationBlocker", "PreparationCancelledError", "PreparationContext", - "PreparationPolicy", "PreparationState", - "RegressionTestResult", "RepairOutcome", "RepairStatus", - "RepositoryTestPlan", "ReproductionSpec", - "ReviewConcern", "SourceIdentity", "SourceIdentityKind", "VerificationDecision", - "VerificationHarness", - "VerificationTarget", "VerifierResult", "build_git_manifest", "build_git_patch", diff --git a/strix/fix/contracts.py b/strix/fix/contracts.py index 47f40ff3b..68f43d325 100644 --- a/strix/fix/contracts.py +++ b/strix/fix/contracts.py @@ -42,11 +42,6 @@ class CheckStatus(StrEnum): SKIPPED = "skipped" -class VerificationTarget(StrEnum): - BASE = "base" - PATCHED = "patched" - - class VerificationDecision(StrEnum): VERIFIED = "verified" REJECTED = "rejected" @@ -152,28 +147,6 @@ class ReproductionSpec(ContractModel): command: CommandSpec | None = None -class RepositoryTestPlan(ContractModel): - """Historical native-test handoff; agent-driven runs record commands directly.""" - - regression_test: CommandSpec - regression_files: list[str] = Field(min_length=1) - unit_tests: list[CommandSpec] = [] - no_unit_tests_reason: str | None = None - - @field_validator("regression_files") - @classmethod - def validate_files(cls, paths: list[str]) -> list[str]: - return list(dict.fromkeys(_validate_relative_path(path) for path in paths)) - - @model_validator(mode="after") - def explain_missing_suite(self) -> RepositoryTestPlan: - if not self.unit_tests and not (self.no_unit_tests_reason or "").strip(): - raise ValueError( - "Provide the existing unit-test commands or explain why no suite exists." - ) - return self - - class ReportedCheck(ContractModel): name: str result: str @@ -230,6 +203,8 @@ class FixPreparationRequestV1(ContractModel): class CheckResult(ContractModel): + """Recorded native shell execution; the reviewer interprets its meaning.""" + name: str argv: list[str] status: CheckStatus @@ -237,90 +212,17 @@ class CheckResult(ContractModel): duration_seconds: float = Field(ge=0) output: str = "" required: bool = True - target: VerificationTarget | None = None - baseline_status: CheckStatus | None = None - baseline_output: str | None = None cwd: str = "." - baseline_exit_code: int | None = None - failure_kind: ( - Literal["environment", "timeout", "check", "source_changed", "harness", "unknown"] | None - ) = None workspace_root: str | None = None - baseline_workspace_root: str | None = None - source_digest: str | None = Field(default=None, pattern=r"^[0-9a-f]{64}$") environment_id: str | None = None - purpose: Literal["quality", "regression", "unit", "security"] = "quality" - tests_passed: int | None = Field(default=None, ge=0) - - -class RegressionTestResult(ContractModel): - """One unchanged test run on both revisions, plus a legitimate-operation check.""" - - name: str - expected_base_failure: str = Field(min_length=1) - harness_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") - base: CheckResult - patched: CheckResult - behavior: CheckResult - - def passed(self) -> bool: - results = (self.base, self.patched, self.behavior) - # Legacy records remain readable. Once provenance is supplied, every leg - # must have it and the environment must be unchanged across the pair. - if any(item.source_digest or item.environment_id for item in results) and ( - not all(item.source_digest and item.environment_id for item in results) - or len({item.environment_id for item in results}) != 1 - or self.patched.source_digest != self.behavior.source_digest - ): - return False - return ( - self.base.target is VerificationTarget.BASE - and self.patched.target is VerificationTarget.PATCHED - and self.behavior.target is VerificationTarget.PATCHED - and self.base.argv == self.patched.argv - and self.base.cwd == self.patched.cwd - and self.base.status is CheckStatus.FAILED - and self.base.exit_code == 1 - and self.base.failure_kind in {None, "check"} - and self.expected_base_failure in self.base.output - and self.patched.status is CheckStatus.PASSED - and self.patched.exit_code == 0 - and self.behavior.status is CheckStatus.PASSED - and self.behavior.exit_code == 0 - ) - - -class VerificationHarness(ContractModel): - path: str - sha256: str = Field(pattern=r"^[0-9a-f]{64}$") - content: str - - -class ReviewConcern(ContractModel): - kind: Literal["repair_needed", "customer_prerequisite", "optional_follow_up"] - summary: str = Field(min_length=1) class VerifierResult(ContractModel): decision: VerificationDecision summary: str - security_invariant_closed: bool = False - reproduction_executed: bool = False - reproduction_summary: str | None = None - sibling_paths_reviewed: list[str] = [] - preserved_behaviors: list[str] = [] gaps: list[str] = [] - security_tests: list[CheckResult] = [] - repairable: bool = False blocker: PreparationBlocker | None = None - regression_tests: list[RegressionTestResult] = [] - harnesses: list[VerificationHarness] = [] - notes: list[str] = [] review_basis: Literal["execution", "code_review"] | None = None - regression_test_valid: bool = False - unit_test_coverage_valid: bool = False - # None preserves historical reviews; new reviewers classify every concern. - concerns: list[ReviewConcern] | None = None source_digest: str | None = Field(default=None, pattern=r"^[0-9a-f]{64}$") turns_used: int = Field(default=0, ge=0) @@ -329,12 +231,9 @@ class RepairOutcome(ContractModel): status: RepairStatus summary: str = Field(min_length=1) gaps: list[str] = [] - reproduction_command: CommandSpec | None = None turns_used: int = Field(default=0, ge=0) blocker: PreparationBlocker | None = None - checks: list[CommandSpec] = [] command_results: list[CheckResult] = [] - test_plan: RepositoryTestPlan | None = None source_digest: str | None = Field(default=None, pattern=r"^[0-9a-f]{64}$") @@ -342,7 +241,6 @@ class FixPreparationAttempt(ContractModel): attempt: int = Field(ge=1) repair: RepairOutcome checks: list[CheckResult] = [] - security_reproduction: CheckResult | None = None verifier: VerifierResult | None = None workspace_digest: str = Field(pattern=r"^[0-9a-f]{64}$") @@ -359,8 +257,7 @@ class FileManifestEntry(ContractModel): class FixPreparationResultV1(ContractModel): version: Literal["1"] = "1" - # Absent on historical records; keep their stronger, paired-proof interpretation. - validation_mode: Literal["paired", "native_tests", "agent_review"] = "paired" + validation_mode: Literal["agent_review"] = "agent_review" prepared_source_digest: str | None = Field(default=None, pattern=r"^[0-9a-f]{64}$") state: PreparationState stop_reason: str @@ -372,16 +269,13 @@ class FixPreparationResultV1(ContractModel): changed_files: list[str] = [] diff_summary: str = "" checks: list[CheckResult] = [] - security_reproduction: CheckResult | None = None verifier: VerifierResult | None = None attempt_history: list[FixPreparationAttempt] = [] gaps: list[str] = [] blocker: PreparationBlocker | None = None - setup_checks: list[CheckResult] = [] attempts: int = Field(default=0, ge=0) elapsed_seconds: float = Field(default=0, ge=0) cost_usd: float | None = Field(default=None, ge=0) - test_plan: RepositoryTestPlan | None = None def candidate_from_legacy_report( diff --git a/strix/fix/evidence.py b/strix/fix/evidence.py deleted file mode 100644 index 3361a4321..000000000 --- a/strix/fix/evidence.py +++ /dev/null @@ -1,93 +0,0 @@ -"""Execution facts shared by local and managed fix preparation.""" - -from __future__ import annotations - -import re -from typing import Literal - -from strix.fix.contracts import CheckResult, CheckStatus, CommandSpec - - -def passed_test_count(output: str) -> int | None: # noqa: PLR0911 - native summary formats - """Recognize native test-runner summaries, not the model's account of a run. - - Counts are optional: runners use different output formats. This is execution - evidence, not an assertion that a test covers the finding; the reviewer checks that. - """ - plain = re.sub(r"\x1b\[[0-9;]*[a-zA-Z]", "", output) - if command_status(0, plain)[0] is CheckStatus.SKIPPED: - return 0 - counts = re.findall(r"\b(\d+)\s+(?:passed|pass)\b", plain, re.IGNORECASE) - counts += re.findall(r"(?:#|\u2139)\s+pass\s+(\d+)\b", plain) - if counts: - return max(int(count) for count in counts) - unittest = re.search(r"Ran (\d+) tests? in [^\n]+\n\s*OK(?: \(skipped=(\d+)\))?", plain) - if unittest: - return max(0, int(unittest[1]) - int(unittest[2] or 0)) - rspec = re.search(r"(\d+) examples?, (\d+) failures?(?:, (\d+) pending)?", plain) - if rspec: - return max(0, int(rspec[1]) - int(rspec[2]) - int(rspec[3] or 0)) - minitest = re.search( - r"(\d+) runs, \d+ assertions, (\d+) failures, (\d+) errors, (\d+) skips", plain - ) - if minitest: - return max(0, int(minitest[1]) - sum(int(minitest[i]) for i in (2, 3, 4))) - if re.search(r"^(?:=+\s*)?[1-9]\d* skipped(?:\s+in [^\n=]+)?(?:\s*=+)?$", plain, re.MULTILINE): - return 0 - go = re.findall(r"^\s*--- PASS: ", plain, re.MULTILINE) - return len(go) if go else None - - -def record_test_execution(result: CheckResult, command: CommandSpec) -> CheckResult: - """Known empty/skipped runs cannot pass; unknown summary formats go to review.""" - count = passed_test_count(result.output) if command.purpose != "quality" else None - update: dict[str, object] = { - "purpose": command.purpose, - "tests_passed": count, - "required": command.required, - "name": command.name, - "argv": command.argv, - "cwd": command.cwd, - } - if command.purpose != "quality" and result.status is CheckStatus.PASSED and count == 0: - update.update( - status=CheckStatus.SKIPPED, - output=result.output + "\nThe runner reported no passing tests.", - ) - return result.model_copy(update=update) - - -def command_status( - exit_code: int, output: str -) -> tuple[CheckStatus, Literal["environment", "check", "harness", "unknown"] | None]: - """Classify launch/collection failures before interpreting an assertion result.""" - if exit_code in {126, 127} or ( - exit_code != 0 - and re.search( - r"(?:command not found|exec: \S+: not found)", - output, - re.IGNORECASE, - ) - ): - return CheckStatus.UNAVAILABLE, "environment" - if exit_code != 0 and re.search( - r"(?:SyntaxError|ReferenceError: require is not defined)", output - ): - return CheckStatus.FAILED, "unknown" - if exit_code != 0 and re.search( - r"(?:No module named |Cannot find module |ERR_MODULE_NOT_FOUND|" - r"ECONNREFUSED|Could not connect to server)", - output, - re.IGNORECASE, - ): - # An incorrect import or a failed service is not evidence that installing - # dependencies is appropriate. Return it to the caller for diagnosis. - return CheckStatus.FAILED, "unknown" - if re.search( - r"(?:Skipping to avoid parser lock|collected 0 items|" - r"No test files found|No tests found, exiting with code 0)", - output, - re.IGNORECASE, - ): - return CheckStatus.SKIPPED, None - return (CheckStatus.PASSED, None) if exit_code == 0 else (CheckStatus.FAILED, "check") diff --git a/strix/fix/prepare.py b/strix/fix/prepare.py index c4de59f24..01a297d01 100644 --- a/strix/fix/prepare.py +++ b/strix/fix/prepare.py @@ -3,14 +3,12 @@ from __future__ import annotations import asyncio -import functools import hashlib import json import logging -import os import subprocess import time -from collections.abc import Awaitable, Callable, Iterable +from collections.abc import Awaitable, Callable from dataclasses import dataclass, field from pathlib import Path from typing import Literal @@ -18,8 +16,6 @@ from typing import Literal from strix.fix.contracts import ( BlockerKind, CheckResult, - CheckStatus, - CommandSpec, FileManifestEntry, FixCandidateV1, FixPreparationAttempt, @@ -32,24 +28,16 @@ from strix.fix.contracts import ( VerificationDecision, VerifierResult, ) -from strix.fix.evidence import command_status class PreparationCancelledError(RuntimeError): pass -CommandRunner = Callable[[Path, CommandSpec], Awaitable[CheckResult]] ManifestBuilder = Callable[[Path], Awaitable[tuple[list[FileManifestEntry], str, str | None]]] CancellationCheck = Callable[[], bool] -@dataclass(slots=True) -class PreparationPolicy: - timeout_seconds: int = 7200 - max_output_chars: int = 20000 - - @dataclass(slots=True) class PreparationContext: request: FixPreparationRequestV1 @@ -61,7 +49,7 @@ class PreparationContext: RepairAgent = Callable[ [PreparationContext, list[CheckResult]], - Awaitable[RepairOutcome | None], + Awaitable[RepairOutcome], ] IndependentVerifier = Callable[ [PreparationContext, list[CheckResult]], @@ -71,127 +59,6 @@ SourceVerifier = Callable[[PreparationContext], Awaitable[bool]] EvidenceReader = Callable[[], Awaitable[list[CheckResult]]] -_COMMAND_ENV_ALLOWLIST = frozenset( - { - "HOME", - "LANG", - "LC_ALL", - "PATH", - "PYTHONHOME", - "PYTHONPATH", - "SYSTEMROOT", - "TEMP", - "TMP", - "TMPDIR", - "VIRTUAL_ENV", - "SystemRoot", - } -) - - -@functools.lru_cache(maxsize=1) -def _network_isolation_prefix() -> tuple[str, ...] | None: - """Return a working ``unshare`` prefix that creates an empty network - namespace, or None when the platform cannot isolate egress.""" - for prefix in (("unshare", "-Urn"), ("unshare", "-n")): - try: - probe = subprocess.run( # noqa: S603 - [*prefix, "true"], - capture_output=True, - timeout=15, - check=False, - ) - except (OSError, subprocess.SubprocessError): - continue - if probe.returncode == 0: - return prefix - return None - - -def _command_environment(credentials_allowed: Iterable[str]) -> dict[str, str]: - allowed = _COMMAND_ENV_ALLOWLIST | set(credentials_allowed) - env = {key: value for key, value in os.environ.items() if key in allowed} - env["PYTHONDONTWRITEBYTECODE"] = "1" - return env - - -async def run_command( - workspace: Path, - command: CommandSpec, - *, - credentials_allowed: Iterable[str] = (), - network_allowed: bool = False, -) -> CheckResult: - started = time.monotonic() - cwd = (workspace / command.cwd).resolve() - if not cwd.is_relative_to(workspace.resolve()) or not cwd.is_dir(): - return CheckResult( - name=command.name, - argv=command.argv, - status=CheckStatus.UNAVAILABLE, - duration_seconds=time.monotonic() - started, - output="The command working directory is unavailable.", - required=command.required, - ) - argv = list(command.argv) - if not network_allowed: - prefix = _network_isolation_prefix() - if prefix is None: - return CheckResult( - name=command.name, - argv=command.argv, - status=CheckStatus.UNAVAILABLE, - duration_seconds=time.monotonic() - started, - output="Network isolation is unavailable, so the command was not run.", - required=command.required, - ) - argv = [*prefix, *argv] - process: asyncio.subprocess.Process | None = None - try: - process = await asyncio.create_subprocess_exec( - *argv, - cwd=cwd, - env=_command_environment(credentials_allowed), - stdout=asyncio.subprocess.PIPE, - stderr=asyncio.subprocess.STDOUT, - ) - output, _ = await asyncio.wait_for(process.communicate(), command.timeout_seconds) - except (FileNotFoundError, PermissionError) as exc: - return CheckResult( - name=command.name, - argv=command.argv, - status=CheckStatus.UNAVAILABLE, - duration_seconds=time.monotonic() - started, - output=str(exc), - required=command.required, - ) - except TimeoutError: - if process is not None: - process.kill() - await process.wait() - return CheckResult( - name=command.name, - argv=command.argv, - status=CheckStatus.FAILED, - duration_seconds=time.monotonic() - started, - output=f"Timed out after {command.timeout_seconds} seconds.", - required=command.required, - ) - status, failure_kind = command_status(process.returncode or 0, output.decode(errors="replace")) - return CheckResult( - name=command.name, - argv=command.argv, - status=status, - exit_code=process.returncode, - duration_seconds=time.monotonic() - started, - output=output.decode(errors="replace")[-20000:], - required=command.required, - failure_kind=failure_kind, - cwd=command.cwd, - workspace_root=str(workspace.resolve()), - ) - - async def build_git_manifest( workspace: Path, ) -> tuple[list[FileManifestEntry], str, str | None]: @@ -361,7 +228,6 @@ def _result( reason: str, started: float, checks: list[CheckResult] | None = None, - reproduction: CheckResult | None = None, verifier: VerifierResult | None = None, gaps: list[str] | None = None, manifest: list[FileManifestEntry] | None = None, @@ -380,7 +246,6 @@ def _result( if context.feedback else None ), - test_plan=context.feedback[-1].repair.test_plan if context.feedback else None, stop_reason=reason, source_identity=context.candidate.source_identity, candidate=context.candidate, @@ -390,7 +255,6 @@ def _result( changed_files=[entry.path for entry in manifest or []], diff_summary=diff_summary, checks=checks or [], - security_reproduction=reproduction, verifier=verifier, attempt_history=attempt_history or [], gaps=gaps or [], @@ -400,15 +264,6 @@ def _result( ) -def _repair_outcome(value: RepairOutcome | None) -> RepairOutcome: - if value is not None: - return value - return RepairOutcome( - status=RepairStatus.COMPLETE, - summary="The repair implementation returned control for independent evaluation.", - ) - - async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup request: FixPreparationRequestV1, workspace: Path, @@ -419,11 +274,9 @@ async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup source_verifier: SourceVerifier = _verify_source, evidence_reader: EvidenceReader | None = None, cancelled: CancellationCheck = lambda: False, - policy: PreparationPolicy | None = None, ) -> FixPreparationResultV1: """Run repair/review conversations; agents own setup, tests and corrections.""" started = time.monotonic() - resolved_policy = policy or PreparationPolicy(timeout_seconds=request.timeout_seconds) context = PreparationContext(request=request, workspace=workspace, candidate=request.candidate) checks: list[CheckResult] = [] verifier: VerifierResult | None = None @@ -453,7 +306,6 @@ async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup attempt_history=context.feedback, blocker=blocker, gaps=gaps, - reproduction=next((c for c in checks if c.purpose == "regression"), None), ) async def execute() -> FixPreparationResultV1: # noqa: PLR0911 - explicit terminal outcomes @@ -482,7 +334,7 @@ async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup workspace_digest=await workspace_digest(workspace), ) context.feedback.append(record) - record.repair = _repair_outcome(await repair(context, checks)) + record.repair = await repair(context, checks) repair_turns += max(1, record.repair.turns_used) checks = await evidence_reader() if evidence_reader else record.repair.command_results record.checks = list(checks) @@ -539,7 +391,7 @@ async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup ) try: - async with asyncio.timeout(resolved_policy.timeout_seconds): + async with asyncio.timeout(request.timeout_seconds): return await execute() except PreparationCancelledError: return await finish(PreparationState.FAILED, "Fix preparation was cancelled.") diff --git a/tests/test_fix_preparation.py b/tests/test_fix_preparation.py index 26988d172..f99c9314b 100644 --- a/tests/test_fix_preparation.py +++ b/tests/test_fix_preparation.py @@ -21,27 +21,21 @@ from strix.fix.contracts import ( FixEdit, FixPreparationRequestV1, PreparationState, - RegressionTestResult, RepairOutcome, RepairStatus, ReproductionSpec, SourceIdentity, SourceIdentityKind, VerificationDecision, - VerificationTarget, VerifierResult, candidate_from_legacy_report, ) -from strix.fix.evidence import command_status, passed_test_count, record_test_execution from strix.fix.locations import AnchorStatus, anchor_location from strix.fix.prepare import ( PreparationContext, - PreparationPolicy, - _network_isolation_prefix, build_git_manifest, build_git_patch, prepare_fix, - run_command, workspace_digest, ) @@ -50,34 +44,6 @@ if TYPE_CHECKING: from pathlib import Path -def _regression() -> RegressionTestResult: - base = CheckResult( - name="authorization", - argv=["python", "regression.py"], - status=CheckStatus.FAILED, - exit_code=1, - duration_seconds=0, - target=VerificationTarget.BASE, - output="unauthorized access allowed", - ) - passed = base.model_copy( - update={ - "status": CheckStatus.PASSED, - "exit_code": 0, - "target": VerificationTarget.PATCHED, - "output": "passed", - } - ) - return RegressionTestResult( - name="authorization", - expected_base_failure="unauthorized access allowed", - harness_sha256="a" * 64, - base=base, - patched=passed, - behavior=passed, - ) - - def _git(workspace: Path, *args: str) -> str: return subprocess.run( # noqa: S603 ["/usr/bin/git", *args], @@ -194,14 +160,7 @@ async def _noop_repair( *context.request.checks, ] digest = await workspace_digest(context.workspace) - results = [] - for command in commands: - result = record_test_execution( - await run_command(context.workspace, command, network_allowed=True), command - ) - results.append( - result.model_copy(update={"source_digest": digest, "environment_id": "test"}) - ) + results = [await _fixture_command(context.workspace, command) for command in commands] return RepairOutcome( status=RepairStatus.COMPLETE, summary="Fixed and tested.", @@ -211,33 +170,36 @@ async def _noop_repair( ) +async def _fixture_command(workspace: Path, command: CommandSpec) -> CheckResult: + """Run this module's synthetic fixture tests, without a production command wrapper.""" + process = await asyncio.create_subprocess_exec( + command.argv[0], + "-B", + *command.argv[1:], + cwd=workspace, + stdout=asyncio.subprocess.PIPE, + stderr=asyncio.subprocess.STDOUT, + ) + output, _ = await process.communicate() + return CheckResult( + name=command.name, + argv=command.argv, + status=CheckStatus.PASSED if process.returncode == 0 else CheckStatus.FAILED, + exit_code=process.returncode, + duration_seconds=0, + output=output.decode(), + ) + + async def _verified( - _context: PreparationContext, + context: PreparationContext, _checks: list[CheckResult], ) -> VerifierResult: return VerifierResult( decision=VerificationDecision.VERIFIED, - source_digest=_context.feedback[-1].repair.source_digest, - summary="The invariant is closed.", - security_invariant_closed=True, + source_digest=context.feedback[-1].repair.source_digest, + summary="The fix addresses the finding.", review_basis="code_review", - regression_test_valid=True, - unit_test_coverage_valid=True, - reproduction_executed=False, - reproduction_summary="The vulnerable input is rejected.", - sibling_paths_reviewed=["app.py"], - preserved_behaviors=["The module compiles."], - regression_tests=[_regression()], - security_tests=[ - CheckResult( - name="security reproduction", - argv=[sys.executable, "-c", "assert True"], - status=CheckStatus.PASSED, - exit_code=0, - duration_seconds=0, - target=VerificationTarget.PATCHED, - ) - ], ) @@ -369,58 +331,6 @@ async def test_build_git_manifest_lists_files_inside_new_directory(tmp_path: Pat assert entry.resulting_sha256 is not None -@pytest.mark.asyncio -async def test_run_command_drops_ambient_credentials( - tmp_path: Path, - monkeypatch: pytest.MonkeyPatch, -) -> None: - monkeypatch.setenv("STRIX_AMBIENT_TOKEN", "hunter2") - command = CommandSpec( - name="env probe", - argv=[ - sys.executable, - "-c", - "import os; print(os.environ.get('STRIX_AMBIENT_TOKEN', ''))", - ], - ) - - sealed = await run_command(tmp_path, command, network_allowed=True) - assert sealed.status is CheckStatus.PASSED - assert "" in sealed.output - - granted = await run_command( - tmp_path, - command, - credentials_allowed=["STRIX_AMBIENT_TOKEN"], - network_allowed=True, - ) - assert granted.status is CheckStatus.PASSED - assert "hunter2" in granted.output - - -@pytest.mark.asyncio -async def test_run_command_blocks_egress_when_network_not_allowed(tmp_path: Path) -> None: - command = CommandSpec( - name="egress probe", - argv=[ - sys.executable, - "-c", - ( - "import socket, sys; s = socket.socket(); s.settimeout(3); " - "sys.exit(0 if s.connect_ex(('1.1.1.1', 53)) == 0 else 1)" - ), - ], - ) - - result = await run_command(tmp_path, command) - - if _network_isolation_prefix() is None: - assert result.status is CheckStatus.UNAVAILABLE - assert "not run" in result.output - else: - assert result.status is CheckStatus.FAILED - - @pytest.mark.asyncio async def test_manifest_patch_includes_untracked_companion_file(tmp_path: Path) -> None: workspace, _ = _workspace(tmp_path) @@ -432,18 +342,6 @@ async def test_manifest_patch_includes_untracked_companion_file(tmp_path: Path) _git(workspace, "diff", "--check") -def test_skipped_check_and_missing_runtime_are_not_passing_evidence() -> None: - assert command_status(0, "Skipping to avoid parser lock")[0] is CheckStatus.SKIPPED - assert command_status(127, "bun: command not found")[0] is CheckStatus.UNAVAILABLE - assert ( - not _regression() - .model_copy( - update={"base": _regression().base.model_copy(update={"failure_kind": "environment"})} - ) - .passed() - ) - - def test_candidate_keeps_full_finding_without_inventing_reproduction() -> None: candidate = candidate_from_legacy_report( { @@ -466,31 +364,6 @@ def test_candidate_keeps_full_finding_without_inventing_reproduction() -> None: assert candidate.reproduction is None -def test_ambiguous_imports_and_syntax_never_authorize_environment_recovery() -> None: - for output in [ - "Cannot find module '/tmp/test/skills/policy.js'", - "No module named 'wrong_repo_path'", - "SyntaxError: invalid syntax", - ]: - status, kind = command_status(1, output) - assert status is CheckStatus.FAILED - assert kind == "unknown" - assert command_status(127, "bun: command not found") == (CheckStatus.UNAVAILABLE, "environment") - - -def test_regression_requires_consistent_execution_provenance() -> None: - regression = _regression() - for leg in (regression.base, regression.patched, regression.behavior): - leg.source_digest = "a" * 64 - leg.environment_id = "execution:0" - assert regression.passed() - regression.base.environment_id = "execution:1" - assert not regression.passed() - regression.base.environment_id = "execution:0" - regression.behavior = regression.behavior.model_copy(update={"source_digest": "b" * 64}) - assert not regression.passed() - - @pytest.mark.asyncio async def test_agent_tests_are_reused_without_controller_execution(tmp_path: Path) -> None: workspace, commit = _workspace(tmp_path) @@ -506,9 +379,8 @@ async def test_agent_tests_are_reused_without_controller_execution(tmp_path: Pat ) assert result.state is PreparationState.READY assert result.validation_mode == "agent_review" - assert result.test_plan is None assert result.checks == outcome.command_results - assert result.checks[0].tests_passed == 1 + assert "Ran 1 test" in result.checks[0].output assert result.prepared_source_digest == result.verifier.source_digest @@ -534,7 +406,7 @@ async def test_review_can_request_more_than_two_repairs(tmp_path: Path) -> None: @pytest.mark.asyncio -@pytest.mark.parametrize("defect", ["missing_unit", "failed", "empty", "stale", "missing_request"]) +@pytest.mark.parametrize("defect", ["missing_unit", "failed", "missing_request"]) async def test_agent_approval_owns_test_evidence_without_controller_retries( tmp_path: Path, defect: str ) -> None: @@ -543,17 +415,14 @@ async def test_agent_approval_owns_test_evidence_without_controller_retries( async def repair(context, checks): result = await _noop_repair(context, checks) if defect == "missing_unit": - result.command_results = [c for c in result.command_results if c.purpose != "unit"] + result.command_results = [ + c for c in result.command_results if c.name != "existing suite" + ] elif defect == "missing_request": - result.command_results = [c for c in result.command_results if c.purpose != "quality"] + result.command_results = [c for c in result.command_results if c.name != "compile"] else: check = result.command_results[0] - if defect == "failed": - check.status, check.exit_code = CheckStatus.FAILED, 1 - elif defect == "empty": - check.tests_passed = 0 - else: - check.source_digest = "a" * 64 + check.status, check.exit_code = CheckStatus.FAILED, 1 return result request = _request(_candidate(commit)) @@ -581,12 +450,7 @@ async def test_incomplete_validation_can_be_finished_by_reviewer(tmp_path: Path) async def review(context, checks): assert checks[1].status is CheckStatus.FAILED command = CommandSpec(name="existing suite", argv=checks[1].argv, purpose="unit") - executed = record_test_execution( - await run_command(workspace, command, network_allowed=True), command - ) - evidence[1] = executed.model_copy( - update={"source_digest": evidence[0].source_digest, "environment_id": "test"} - ) + evidence[1] = await _fixture_command(workspace, command) return await _verified(context, checks) async def read_evidence(): @@ -631,12 +495,13 @@ async def test_interruptions_preserve_partial_patch_without_approval( raise AssertionError("Stopped repair must not be approved") result = await prepare_fix( - _request(_candidate(commit)), + _request(_candidate(commit)).model_copy( + update={"timeout_seconds": 1 if stop == "timeout" else 30} + ), workspace, repair=repair, verify=review, cancelled=lambda: cancel, - policy=PreparationPolicy(timeout_seconds=1 if stop == "timeout" else 30), ) assert result.state in {PreparationState.BLOCKED, PreparationState.FAILED} assert result.final_file_manifest @@ -686,24 +551,6 @@ async def test_empty_deliverable_does_not_start_another_repair(tmp_path: Path) - assert "without a deliverable patch" in result.stop_reason -@pytest.mark.parametrize( - ("output", "expected"), - [ - ("Tests: 2 passed, 2 total", 2), - ("2 pass\n0 fail", 2), - ("# tests 2\n# pass 2\n# fail 0", 2), - ("Ran 3 tests in 0.1s\n\nOK (skipped=1)", 2), - ("3 examples, 0 failures, 1 pending", 2), - ("--- PASS: TestSafe (0.00s)", 1), - ("Ran 2 tests in 0.1s\n\nOK (skipped=2)", 0), - ("No tests found", None), - ], -) -def test_runner_summaries_record_actual_execution(output: str, expected: int | None) -> None: - - assert passed_test_count(output) == expected - - def test_new_command_metadata_does_not_change_existing_finding_digest(tmp_path: Path) -> None: _root, commit = _workspace(tmp_path) @@ -715,21 +562,3 @@ def test_new_command_metadata_does_not_change_existing_finding_digest(tmp_path: json.dumps(payload, sort_keys=True, separators=(",", ":")).encode() ).hexdigest() assert candidate.digest() == previous - - -def test_unknown_runner_format_preserves_exit_result_for_independent_review() -> None: - - command = CommandSpec(name="custom runner", argv=["./tests/run"], purpose="unit") - result = record_test_execution( - CheckResult( - name=command.name, - argv=command.argv, - status=CheckStatus.PASSED, - exit_code=0, - duration_seconds=1, - output="All project assertions completed successfully.", - ), - command, - ) - assert result.status is CheckStatus.PASSED - assert result.tests_passed is None