mirror of
https://github.com/usestrix/strix.git
synced 2026-10-02 02:13:43 +00:00
Remove retired fix execution and proof machinery
This commit is contained in:
parent
348fbf2608
commit
5badb2d541
5 changed files with 43 additions and 573 deletions
|
|
@ -15,24 +15,18 @@ from strix.fix.contracts import (
|
|||
FixPreparationResultV1,
|
||||
PreparationBlocker,
|
||||
PreparationState,
|
||||
RegressionTestResult,
|
||||
RepairOutcome,
|
||||
RepairStatus,
|
||||
RepositoryTestPlan,
|
||||
ReproductionSpec,
|
||||
ReviewConcern,
|
||||
SourceIdentity,
|
||||
SourceIdentityKind,
|
||||
VerificationDecision,
|
||||
VerificationHarness,
|
||||
VerificationTarget,
|
||||
VerifierResult,
|
||||
candidate_from_legacy_report,
|
||||
)
|
||||
from strix.fix.prepare import (
|
||||
PreparationCancelledError,
|
||||
PreparationContext,
|
||||
PreparationPolicy,
|
||||
build_git_manifest,
|
||||
build_git_patch,
|
||||
prepare_fix,
|
||||
|
|
@ -56,19 +50,13 @@ __all__ = [
|
|||
"PreparationBlocker",
|
||||
"PreparationCancelledError",
|
||||
"PreparationContext",
|
||||
"PreparationPolicy",
|
||||
"PreparationState",
|
||||
"RegressionTestResult",
|
||||
"RepairOutcome",
|
||||
"RepairStatus",
|
||||
"RepositoryTestPlan",
|
||||
"ReproductionSpec",
|
||||
"ReviewConcern",
|
||||
"SourceIdentity",
|
||||
"SourceIdentityKind",
|
||||
"VerificationDecision",
|
||||
"VerificationHarness",
|
||||
"VerificationTarget",
|
||||
"VerifierResult",
|
||||
"build_git_manifest",
|
||||
"build_git_patch",
|
||||
|
|
|
|||
|
|
@ -42,11 +42,6 @@ class CheckStatus(StrEnum):
|
|||
SKIPPED = "skipped"
|
||||
|
||||
|
||||
class VerificationTarget(StrEnum):
|
||||
BASE = "base"
|
||||
PATCHED = "patched"
|
||||
|
||||
|
||||
class VerificationDecision(StrEnum):
|
||||
VERIFIED = "verified"
|
||||
REJECTED = "rejected"
|
||||
|
|
@ -152,28 +147,6 @@ class ReproductionSpec(ContractModel):
|
|||
command: CommandSpec | None = None
|
||||
|
||||
|
||||
class RepositoryTestPlan(ContractModel):
|
||||
"""Historical native-test handoff; agent-driven runs record commands directly."""
|
||||
|
||||
regression_test: CommandSpec
|
||||
regression_files: list[str] = Field(min_length=1)
|
||||
unit_tests: list[CommandSpec] = []
|
||||
no_unit_tests_reason: str | None = None
|
||||
|
||||
@field_validator("regression_files")
|
||||
@classmethod
|
||||
def validate_files(cls, paths: list[str]) -> list[str]:
|
||||
return list(dict.fromkeys(_validate_relative_path(path) for path in paths))
|
||||
|
||||
@model_validator(mode="after")
|
||||
def explain_missing_suite(self) -> RepositoryTestPlan:
|
||||
if not self.unit_tests and not (self.no_unit_tests_reason or "").strip():
|
||||
raise ValueError(
|
||||
"Provide the existing unit-test commands or explain why no suite exists."
|
||||
)
|
||||
return self
|
||||
|
||||
|
||||
class ReportedCheck(ContractModel):
|
||||
name: str
|
||||
result: str
|
||||
|
|
@ -230,6 +203,8 @@ class FixPreparationRequestV1(ContractModel):
|
|||
|
||||
|
||||
class CheckResult(ContractModel):
|
||||
"""Recorded native shell execution; the reviewer interprets its meaning."""
|
||||
|
||||
name: str
|
||||
argv: list[str]
|
||||
status: CheckStatus
|
||||
|
|
@ -237,90 +212,17 @@ class CheckResult(ContractModel):
|
|||
duration_seconds: float = Field(ge=0)
|
||||
output: str = ""
|
||||
required: bool = True
|
||||
target: VerificationTarget | None = None
|
||||
baseline_status: CheckStatus | None = None
|
||||
baseline_output: str | None = None
|
||||
cwd: str = "."
|
||||
baseline_exit_code: int | None = None
|
||||
failure_kind: (
|
||||
Literal["environment", "timeout", "check", "source_changed", "harness", "unknown"] | None
|
||||
) = None
|
||||
workspace_root: str | None = None
|
||||
baseline_workspace_root: str | None = None
|
||||
source_digest: str | None = Field(default=None, pattern=r"^[0-9a-f]{64}$")
|
||||
environment_id: str | None = None
|
||||
purpose: Literal["quality", "regression", "unit", "security"] = "quality"
|
||||
tests_passed: int | None = Field(default=None, ge=0)
|
||||
|
||||
|
||||
class RegressionTestResult(ContractModel):
|
||||
"""One unchanged test run on both revisions, plus a legitimate-operation check."""
|
||||
|
||||
name: str
|
||||
expected_base_failure: str = Field(min_length=1)
|
||||
harness_sha256: str = Field(pattern=r"^[0-9a-f]{64}$")
|
||||
base: CheckResult
|
||||
patched: CheckResult
|
||||
behavior: CheckResult
|
||||
|
||||
def passed(self) -> bool:
|
||||
results = (self.base, self.patched, self.behavior)
|
||||
# Legacy records remain readable. Once provenance is supplied, every leg
|
||||
# must have it and the environment must be unchanged across the pair.
|
||||
if any(item.source_digest or item.environment_id for item in results) and (
|
||||
not all(item.source_digest and item.environment_id for item in results)
|
||||
or len({item.environment_id for item in results}) != 1
|
||||
or self.patched.source_digest != self.behavior.source_digest
|
||||
):
|
||||
return False
|
||||
return (
|
||||
self.base.target is VerificationTarget.BASE
|
||||
and self.patched.target is VerificationTarget.PATCHED
|
||||
and self.behavior.target is VerificationTarget.PATCHED
|
||||
and self.base.argv == self.patched.argv
|
||||
and self.base.cwd == self.patched.cwd
|
||||
and self.base.status is CheckStatus.FAILED
|
||||
and self.base.exit_code == 1
|
||||
and self.base.failure_kind in {None, "check"}
|
||||
and self.expected_base_failure in self.base.output
|
||||
and self.patched.status is CheckStatus.PASSED
|
||||
and self.patched.exit_code == 0
|
||||
and self.behavior.status is CheckStatus.PASSED
|
||||
and self.behavior.exit_code == 0
|
||||
)
|
||||
|
||||
|
||||
class VerificationHarness(ContractModel):
|
||||
path: str
|
||||
sha256: str = Field(pattern=r"^[0-9a-f]{64}$")
|
||||
content: str
|
||||
|
||||
|
||||
class ReviewConcern(ContractModel):
|
||||
kind: Literal["repair_needed", "customer_prerequisite", "optional_follow_up"]
|
||||
summary: str = Field(min_length=1)
|
||||
|
||||
|
||||
class VerifierResult(ContractModel):
|
||||
decision: VerificationDecision
|
||||
summary: str
|
||||
security_invariant_closed: bool = False
|
||||
reproduction_executed: bool = False
|
||||
reproduction_summary: str | None = None
|
||||
sibling_paths_reviewed: list[str] = []
|
||||
preserved_behaviors: list[str] = []
|
||||
gaps: list[str] = []
|
||||
security_tests: list[CheckResult] = []
|
||||
repairable: bool = False
|
||||
blocker: PreparationBlocker | None = None
|
||||
regression_tests: list[RegressionTestResult] = []
|
||||
harnesses: list[VerificationHarness] = []
|
||||
notes: list[str] = []
|
||||
review_basis: Literal["execution", "code_review"] | None = None
|
||||
regression_test_valid: bool = False
|
||||
unit_test_coverage_valid: bool = False
|
||||
# None preserves historical reviews; new reviewers classify every concern.
|
||||
concerns: list[ReviewConcern] | None = None
|
||||
source_digest: str | None = Field(default=None, pattern=r"^[0-9a-f]{64}$")
|
||||
turns_used: int = Field(default=0, ge=0)
|
||||
|
||||
|
|
@ -329,12 +231,9 @@ class RepairOutcome(ContractModel):
|
|||
status: RepairStatus
|
||||
summary: str = Field(min_length=1)
|
||||
gaps: list[str] = []
|
||||
reproduction_command: CommandSpec | None = None
|
||||
turns_used: int = Field(default=0, ge=0)
|
||||
blocker: PreparationBlocker | None = None
|
||||
checks: list[CommandSpec] = []
|
||||
command_results: list[CheckResult] = []
|
||||
test_plan: RepositoryTestPlan | None = None
|
||||
source_digest: str | None = Field(default=None, pattern=r"^[0-9a-f]{64}$")
|
||||
|
||||
|
||||
|
|
@ -342,7 +241,6 @@ class FixPreparationAttempt(ContractModel):
|
|||
attempt: int = Field(ge=1)
|
||||
repair: RepairOutcome
|
||||
checks: list[CheckResult] = []
|
||||
security_reproduction: CheckResult | None = None
|
||||
verifier: VerifierResult | None = None
|
||||
workspace_digest: str = Field(pattern=r"^[0-9a-f]{64}$")
|
||||
|
||||
|
|
@ -359,8 +257,7 @@ class FileManifestEntry(ContractModel):
|
|||
|
||||
class FixPreparationResultV1(ContractModel):
|
||||
version: Literal["1"] = "1"
|
||||
# Absent on historical records; keep their stronger, paired-proof interpretation.
|
||||
validation_mode: Literal["paired", "native_tests", "agent_review"] = "paired"
|
||||
validation_mode: Literal["agent_review"] = "agent_review"
|
||||
prepared_source_digest: str | None = Field(default=None, pattern=r"^[0-9a-f]{64}$")
|
||||
state: PreparationState
|
||||
stop_reason: str
|
||||
|
|
@ -372,16 +269,13 @@ class FixPreparationResultV1(ContractModel):
|
|||
changed_files: list[str] = []
|
||||
diff_summary: str = ""
|
||||
checks: list[CheckResult] = []
|
||||
security_reproduction: CheckResult | None = None
|
||||
verifier: VerifierResult | None = None
|
||||
attempt_history: list[FixPreparationAttempt] = []
|
||||
gaps: list[str] = []
|
||||
blocker: PreparationBlocker | None = None
|
||||
setup_checks: list[CheckResult] = []
|
||||
attempts: int = Field(default=0, ge=0)
|
||||
elapsed_seconds: float = Field(default=0, ge=0)
|
||||
cost_usd: float | None = Field(default=None, ge=0)
|
||||
test_plan: RepositoryTestPlan | None = None
|
||||
|
||||
|
||||
def candidate_from_legacy_report(
|
||||
|
|
|
|||
|
|
@ -1,93 +0,0 @@
|
|||
"""Execution facts shared by local and managed fix preparation."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Literal
|
||||
|
||||
from strix.fix.contracts import CheckResult, CheckStatus, CommandSpec
|
||||
|
||||
|
||||
def passed_test_count(output: str) -> int | None: # noqa: PLR0911 - native summary formats
|
||||
"""Recognize native test-runner summaries, not the model's account of a run.
|
||||
|
||||
Counts are optional: runners use different output formats. This is execution
|
||||
evidence, not an assertion that a test covers the finding; the reviewer checks that.
|
||||
"""
|
||||
plain = re.sub(r"\x1b\[[0-9;]*[a-zA-Z]", "", output)
|
||||
if command_status(0, plain)[0] is CheckStatus.SKIPPED:
|
||||
return 0
|
||||
counts = re.findall(r"\b(\d+)\s+(?:passed|pass)\b", plain, re.IGNORECASE)
|
||||
counts += re.findall(r"(?:#|\u2139)\s+pass\s+(\d+)\b", plain)
|
||||
if counts:
|
||||
return max(int(count) for count in counts)
|
||||
unittest = re.search(r"Ran (\d+) tests? in [^\n]+\n\s*OK(?: \(skipped=(\d+)\))?", plain)
|
||||
if unittest:
|
||||
return max(0, int(unittest[1]) - int(unittest[2] or 0))
|
||||
rspec = re.search(r"(\d+) examples?, (\d+) failures?(?:, (\d+) pending)?", plain)
|
||||
if rspec:
|
||||
return max(0, int(rspec[1]) - int(rspec[2]) - int(rspec[3] or 0))
|
||||
minitest = re.search(
|
||||
r"(\d+) runs, \d+ assertions, (\d+) failures, (\d+) errors, (\d+) skips", plain
|
||||
)
|
||||
if minitest:
|
||||
return max(0, int(minitest[1]) - sum(int(minitest[i]) for i in (2, 3, 4)))
|
||||
if re.search(r"^(?:=+\s*)?[1-9]\d* skipped(?:\s+in [^\n=]+)?(?:\s*=+)?$", plain, re.MULTILINE):
|
||||
return 0
|
||||
go = re.findall(r"^\s*--- PASS: ", plain, re.MULTILINE)
|
||||
return len(go) if go else None
|
||||
|
||||
|
||||
def record_test_execution(result: CheckResult, command: CommandSpec) -> CheckResult:
|
||||
"""Known empty/skipped runs cannot pass; unknown summary formats go to review."""
|
||||
count = passed_test_count(result.output) if command.purpose != "quality" else None
|
||||
update: dict[str, object] = {
|
||||
"purpose": command.purpose,
|
||||
"tests_passed": count,
|
||||
"required": command.required,
|
||||
"name": command.name,
|
||||
"argv": command.argv,
|
||||
"cwd": command.cwd,
|
||||
}
|
||||
if command.purpose != "quality" and result.status is CheckStatus.PASSED and count == 0:
|
||||
update.update(
|
||||
status=CheckStatus.SKIPPED,
|
||||
output=result.output + "\nThe runner reported no passing tests.",
|
||||
)
|
||||
return result.model_copy(update=update)
|
||||
|
||||
|
||||
def command_status(
|
||||
exit_code: int, output: str
|
||||
) -> tuple[CheckStatus, Literal["environment", "check", "harness", "unknown"] | None]:
|
||||
"""Classify launch/collection failures before interpreting an assertion result."""
|
||||
if exit_code in {126, 127} or (
|
||||
exit_code != 0
|
||||
and re.search(
|
||||
r"(?:command not found|exec: \S+: not found)",
|
||||
output,
|
||||
re.IGNORECASE,
|
||||
)
|
||||
):
|
||||
return CheckStatus.UNAVAILABLE, "environment"
|
||||
if exit_code != 0 and re.search(
|
||||
r"(?:SyntaxError|ReferenceError: require is not defined)", output
|
||||
):
|
||||
return CheckStatus.FAILED, "unknown"
|
||||
if exit_code != 0 and re.search(
|
||||
r"(?:No module named |Cannot find module |ERR_MODULE_NOT_FOUND|"
|
||||
r"ECONNREFUSED|Could not connect to server)",
|
||||
output,
|
||||
re.IGNORECASE,
|
||||
):
|
||||
# An incorrect import or a failed service is not evidence that installing
|
||||
# dependencies is appropriate. Return it to the caller for diagnosis.
|
||||
return CheckStatus.FAILED, "unknown"
|
||||
if re.search(
|
||||
r"(?:Skipping to avoid parser lock|collected 0 items|"
|
||||
r"No test files found|No tests found, exiting with code 0)",
|
||||
output,
|
||||
re.IGNORECASE,
|
||||
):
|
||||
return CheckStatus.SKIPPED, None
|
||||
return (CheckStatus.PASSED, None) if exit_code == 0 else (CheckStatus.FAILED, "check")
|
||||
|
|
@ -3,14 +3,12 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import functools
|
||||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
import time
|
||||
from collections.abc import Awaitable, Callable, Iterable
|
||||
from collections.abc import Awaitable, Callable
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Literal
|
||||
|
|
@ -18,8 +16,6 @@ from typing import Literal
|
|||
from strix.fix.contracts import (
|
||||
BlockerKind,
|
||||
CheckResult,
|
||||
CheckStatus,
|
||||
CommandSpec,
|
||||
FileManifestEntry,
|
||||
FixCandidateV1,
|
||||
FixPreparationAttempt,
|
||||
|
|
@ -32,24 +28,16 @@ from strix.fix.contracts import (
|
|||
VerificationDecision,
|
||||
VerifierResult,
|
||||
)
|
||||
from strix.fix.evidence import command_status
|
||||
|
||||
|
||||
class PreparationCancelledError(RuntimeError):
|
||||
pass
|
||||
|
||||
|
||||
CommandRunner = Callable[[Path, CommandSpec], Awaitable[CheckResult]]
|
||||
ManifestBuilder = Callable[[Path], Awaitable[tuple[list[FileManifestEntry], str, str | None]]]
|
||||
CancellationCheck = Callable[[], bool]
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
class PreparationPolicy:
|
||||
timeout_seconds: int = 7200
|
||||
max_output_chars: int = 20000
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
class PreparationContext:
|
||||
request: FixPreparationRequestV1
|
||||
|
|
@ -61,7 +49,7 @@ class PreparationContext:
|
|||
|
||||
RepairAgent = Callable[
|
||||
[PreparationContext, list[CheckResult]],
|
||||
Awaitable[RepairOutcome | None],
|
||||
Awaitable[RepairOutcome],
|
||||
]
|
||||
IndependentVerifier = Callable[
|
||||
[PreparationContext, list[CheckResult]],
|
||||
|
|
@ -71,127 +59,6 @@ SourceVerifier = Callable[[PreparationContext], Awaitable[bool]]
|
|||
EvidenceReader = Callable[[], Awaitable[list[CheckResult]]]
|
||||
|
||||
|
||||
_COMMAND_ENV_ALLOWLIST = frozenset(
|
||||
{
|
||||
"HOME",
|
||||
"LANG",
|
||||
"LC_ALL",
|
||||
"PATH",
|
||||
"PYTHONHOME",
|
||||
"PYTHONPATH",
|
||||
"SYSTEMROOT",
|
||||
"TEMP",
|
||||
"TMP",
|
||||
"TMPDIR",
|
||||
"VIRTUAL_ENV",
|
||||
"SystemRoot",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
@functools.lru_cache(maxsize=1)
|
||||
def _network_isolation_prefix() -> tuple[str, ...] | None:
|
||||
"""Return a working ``unshare`` prefix that creates an empty network
|
||||
namespace, or None when the platform cannot isolate egress."""
|
||||
for prefix in (("unshare", "-Urn"), ("unshare", "-n")):
|
||||
try:
|
||||
probe = subprocess.run( # noqa: S603
|
||||
[*prefix, "true"],
|
||||
capture_output=True,
|
||||
timeout=15,
|
||||
check=False,
|
||||
)
|
||||
except (OSError, subprocess.SubprocessError):
|
||||
continue
|
||||
if probe.returncode == 0:
|
||||
return prefix
|
||||
return None
|
||||
|
||||
|
||||
def _command_environment(credentials_allowed: Iterable[str]) -> dict[str, str]:
|
||||
allowed = _COMMAND_ENV_ALLOWLIST | set(credentials_allowed)
|
||||
env = {key: value for key, value in os.environ.items() if key in allowed}
|
||||
env["PYTHONDONTWRITEBYTECODE"] = "1"
|
||||
return env
|
||||
|
||||
|
||||
async def run_command(
|
||||
workspace: Path,
|
||||
command: CommandSpec,
|
||||
*,
|
||||
credentials_allowed: Iterable[str] = (),
|
||||
network_allowed: bool = False,
|
||||
) -> CheckResult:
|
||||
started = time.monotonic()
|
||||
cwd = (workspace / command.cwd).resolve()
|
||||
if not cwd.is_relative_to(workspace.resolve()) or not cwd.is_dir():
|
||||
return CheckResult(
|
||||
name=command.name,
|
||||
argv=command.argv,
|
||||
status=CheckStatus.UNAVAILABLE,
|
||||
duration_seconds=time.monotonic() - started,
|
||||
output="The command working directory is unavailable.",
|
||||
required=command.required,
|
||||
)
|
||||
argv = list(command.argv)
|
||||
if not network_allowed:
|
||||
prefix = _network_isolation_prefix()
|
||||
if prefix is None:
|
||||
return CheckResult(
|
||||
name=command.name,
|
||||
argv=command.argv,
|
||||
status=CheckStatus.UNAVAILABLE,
|
||||
duration_seconds=time.monotonic() - started,
|
||||
output="Network isolation is unavailable, so the command was not run.",
|
||||
required=command.required,
|
||||
)
|
||||
argv = [*prefix, *argv]
|
||||
process: asyncio.subprocess.Process | None = None
|
||||
try:
|
||||
process = await asyncio.create_subprocess_exec(
|
||||
*argv,
|
||||
cwd=cwd,
|
||||
env=_command_environment(credentials_allowed),
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.STDOUT,
|
||||
)
|
||||
output, _ = await asyncio.wait_for(process.communicate(), command.timeout_seconds)
|
||||
except (FileNotFoundError, PermissionError) as exc:
|
||||
return CheckResult(
|
||||
name=command.name,
|
||||
argv=command.argv,
|
||||
status=CheckStatus.UNAVAILABLE,
|
||||
duration_seconds=time.monotonic() - started,
|
||||
output=str(exc),
|
||||
required=command.required,
|
||||
)
|
||||
except TimeoutError:
|
||||
if process is not None:
|
||||
process.kill()
|
||||
await process.wait()
|
||||
return CheckResult(
|
||||
name=command.name,
|
||||
argv=command.argv,
|
||||
status=CheckStatus.FAILED,
|
||||
duration_seconds=time.monotonic() - started,
|
||||
output=f"Timed out after {command.timeout_seconds} seconds.",
|
||||
required=command.required,
|
||||
)
|
||||
status, failure_kind = command_status(process.returncode or 0, output.decode(errors="replace"))
|
||||
return CheckResult(
|
||||
name=command.name,
|
||||
argv=command.argv,
|
||||
status=status,
|
||||
exit_code=process.returncode,
|
||||
duration_seconds=time.monotonic() - started,
|
||||
output=output.decode(errors="replace")[-20000:],
|
||||
required=command.required,
|
||||
failure_kind=failure_kind,
|
||||
cwd=command.cwd,
|
||||
workspace_root=str(workspace.resolve()),
|
||||
)
|
||||
|
||||
|
||||
async def build_git_manifest(
|
||||
workspace: Path,
|
||||
) -> tuple[list[FileManifestEntry], str, str | None]:
|
||||
|
|
@ -361,7 +228,6 @@ def _result(
|
|||
reason: str,
|
||||
started: float,
|
||||
checks: list[CheckResult] | None = None,
|
||||
reproduction: CheckResult | None = None,
|
||||
verifier: VerifierResult | None = None,
|
||||
gaps: list[str] | None = None,
|
||||
manifest: list[FileManifestEntry] | None = None,
|
||||
|
|
@ -380,7 +246,6 @@ def _result(
|
|||
if context.feedback
|
||||
else None
|
||||
),
|
||||
test_plan=context.feedback[-1].repair.test_plan if context.feedback else None,
|
||||
stop_reason=reason,
|
||||
source_identity=context.candidate.source_identity,
|
||||
candidate=context.candidate,
|
||||
|
|
@ -390,7 +255,6 @@ def _result(
|
|||
changed_files=[entry.path for entry in manifest or []],
|
||||
diff_summary=diff_summary,
|
||||
checks=checks or [],
|
||||
security_reproduction=reproduction,
|
||||
verifier=verifier,
|
||||
attempt_history=attempt_history or [],
|
||||
gaps=gaps or [],
|
||||
|
|
@ -400,15 +264,6 @@ def _result(
|
|||
)
|
||||
|
||||
|
||||
def _repair_outcome(value: RepairOutcome | None) -> RepairOutcome:
|
||||
if value is not None:
|
||||
return value
|
||||
return RepairOutcome(
|
||||
status=RepairStatus.COMPLETE,
|
||||
summary="The repair implementation returned control for independent evaluation.",
|
||||
)
|
||||
|
||||
|
||||
async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup
|
||||
request: FixPreparationRequestV1,
|
||||
workspace: Path,
|
||||
|
|
@ -419,11 +274,9 @@ async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup
|
|||
source_verifier: SourceVerifier = _verify_source,
|
||||
evidence_reader: EvidenceReader | None = None,
|
||||
cancelled: CancellationCheck = lambda: False,
|
||||
policy: PreparationPolicy | None = None,
|
||||
) -> FixPreparationResultV1:
|
||||
"""Run repair/review conversations; agents own setup, tests and corrections."""
|
||||
started = time.monotonic()
|
||||
resolved_policy = policy or PreparationPolicy(timeout_seconds=request.timeout_seconds)
|
||||
context = PreparationContext(request=request, workspace=workspace, candidate=request.candidate)
|
||||
checks: list[CheckResult] = []
|
||||
verifier: VerifierResult | None = None
|
||||
|
|
@ -453,7 +306,6 @@ async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup
|
|||
attempt_history=context.feedback,
|
||||
blocker=blocker,
|
||||
gaps=gaps,
|
||||
reproduction=next((c for c in checks if c.purpose == "regression"), None),
|
||||
)
|
||||
|
||||
async def execute() -> FixPreparationResultV1: # noqa: PLR0911 - explicit terminal outcomes
|
||||
|
|
@ -482,7 +334,7 @@ async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup
|
|||
workspace_digest=await workspace_digest(workspace),
|
||||
)
|
||||
context.feedback.append(record)
|
||||
record.repair = _repair_outcome(await repair(context, checks))
|
||||
record.repair = await repair(context, checks)
|
||||
repair_turns += max(1, record.repair.turns_used)
|
||||
checks = await evidence_reader() if evidence_reader else record.repair.command_results
|
||||
record.checks = list(checks)
|
||||
|
|
@ -539,7 +391,7 @@ async def prepare_fix( # noqa: PLR0915 - thin orchestration and cleanup
|
|||
)
|
||||
|
||||
try:
|
||||
async with asyncio.timeout(resolved_policy.timeout_seconds):
|
||||
async with asyncio.timeout(request.timeout_seconds):
|
||||
return await execute()
|
||||
except PreparationCancelledError:
|
||||
return await finish(PreparationState.FAILED, "Fix preparation was cancelled.")
|
||||
|
|
|
|||
|
|
@ -21,27 +21,21 @@ from strix.fix.contracts import (
|
|||
FixEdit,
|
||||
FixPreparationRequestV1,
|
||||
PreparationState,
|
||||
RegressionTestResult,
|
||||
RepairOutcome,
|
||||
RepairStatus,
|
||||
ReproductionSpec,
|
||||
SourceIdentity,
|
||||
SourceIdentityKind,
|
||||
VerificationDecision,
|
||||
VerificationTarget,
|
||||
VerifierResult,
|
||||
candidate_from_legacy_report,
|
||||
)
|
||||
from strix.fix.evidence import command_status, passed_test_count, record_test_execution
|
||||
from strix.fix.locations import AnchorStatus, anchor_location
|
||||
from strix.fix.prepare import (
|
||||
PreparationContext,
|
||||
PreparationPolicy,
|
||||
_network_isolation_prefix,
|
||||
build_git_manifest,
|
||||
build_git_patch,
|
||||
prepare_fix,
|
||||
run_command,
|
||||
workspace_digest,
|
||||
)
|
||||
|
||||
|
|
@ -50,34 +44,6 @@ if TYPE_CHECKING:
|
|||
from pathlib import Path
|
||||
|
||||
|
||||
def _regression() -> RegressionTestResult:
|
||||
base = CheckResult(
|
||||
name="authorization",
|
||||
argv=["python", "regression.py"],
|
||||
status=CheckStatus.FAILED,
|
||||
exit_code=1,
|
||||
duration_seconds=0,
|
||||
target=VerificationTarget.BASE,
|
||||
output="unauthorized access allowed",
|
||||
)
|
||||
passed = base.model_copy(
|
||||
update={
|
||||
"status": CheckStatus.PASSED,
|
||||
"exit_code": 0,
|
||||
"target": VerificationTarget.PATCHED,
|
||||
"output": "passed",
|
||||
}
|
||||
)
|
||||
return RegressionTestResult(
|
||||
name="authorization",
|
||||
expected_base_failure="unauthorized access allowed",
|
||||
harness_sha256="a" * 64,
|
||||
base=base,
|
||||
patched=passed,
|
||||
behavior=passed,
|
||||
)
|
||||
|
||||
|
||||
def _git(workspace: Path, *args: str) -> str:
|
||||
return subprocess.run( # noqa: S603
|
||||
["/usr/bin/git", *args],
|
||||
|
|
@ -194,14 +160,7 @@ async def _noop_repair(
|
|||
*context.request.checks,
|
||||
]
|
||||
digest = await workspace_digest(context.workspace)
|
||||
results = []
|
||||
for command in commands:
|
||||
result = record_test_execution(
|
||||
await run_command(context.workspace, command, network_allowed=True), command
|
||||
)
|
||||
results.append(
|
||||
result.model_copy(update={"source_digest": digest, "environment_id": "test"})
|
||||
)
|
||||
results = [await _fixture_command(context.workspace, command) for command in commands]
|
||||
return RepairOutcome(
|
||||
status=RepairStatus.COMPLETE,
|
||||
summary="Fixed and tested.",
|
||||
|
|
@ -211,33 +170,36 @@ async def _noop_repair(
|
|||
)
|
||||
|
||||
|
||||
async def _fixture_command(workspace: Path, command: CommandSpec) -> CheckResult:
|
||||
"""Run this module's synthetic fixture tests, without a production command wrapper."""
|
||||
process = await asyncio.create_subprocess_exec(
|
||||
command.argv[0],
|
||||
"-B",
|
||||
*command.argv[1:],
|
||||
cwd=workspace,
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.STDOUT,
|
||||
)
|
||||
output, _ = await process.communicate()
|
||||
return CheckResult(
|
||||
name=command.name,
|
||||
argv=command.argv,
|
||||
status=CheckStatus.PASSED if process.returncode == 0 else CheckStatus.FAILED,
|
||||
exit_code=process.returncode,
|
||||
duration_seconds=0,
|
||||
output=output.decode(),
|
||||
)
|
||||
|
||||
|
||||
async def _verified(
|
||||
_context: PreparationContext,
|
||||
context: PreparationContext,
|
||||
_checks: list[CheckResult],
|
||||
) -> VerifierResult:
|
||||
return VerifierResult(
|
||||
decision=VerificationDecision.VERIFIED,
|
||||
source_digest=_context.feedback[-1].repair.source_digest,
|
||||
summary="The invariant is closed.",
|
||||
security_invariant_closed=True,
|
||||
source_digest=context.feedback[-1].repair.source_digest,
|
||||
summary="The fix addresses the finding.",
|
||||
review_basis="code_review",
|
||||
regression_test_valid=True,
|
||||
unit_test_coverage_valid=True,
|
||||
reproduction_executed=False,
|
||||
reproduction_summary="The vulnerable input is rejected.",
|
||||
sibling_paths_reviewed=["app.py"],
|
||||
preserved_behaviors=["The module compiles."],
|
||||
regression_tests=[_regression()],
|
||||
security_tests=[
|
||||
CheckResult(
|
||||
name="security reproduction",
|
||||
argv=[sys.executable, "-c", "assert True"],
|
||||
status=CheckStatus.PASSED,
|
||||
exit_code=0,
|
||||
duration_seconds=0,
|
||||
target=VerificationTarget.PATCHED,
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -369,58 +331,6 @@ async def test_build_git_manifest_lists_files_inside_new_directory(tmp_path: Pat
|
|||
assert entry.resulting_sha256 is not None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_run_command_drops_ambient_credentials(
|
||||
tmp_path: Path,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
monkeypatch.setenv("STRIX_AMBIENT_TOKEN", "hunter2")
|
||||
command = CommandSpec(
|
||||
name="env probe",
|
||||
argv=[
|
||||
sys.executable,
|
||||
"-c",
|
||||
"import os; print(os.environ.get('STRIX_AMBIENT_TOKEN', '<absent>'))",
|
||||
],
|
||||
)
|
||||
|
||||
sealed = await run_command(tmp_path, command, network_allowed=True)
|
||||
assert sealed.status is CheckStatus.PASSED
|
||||
assert "<absent>" in sealed.output
|
||||
|
||||
granted = await run_command(
|
||||
tmp_path,
|
||||
command,
|
||||
credentials_allowed=["STRIX_AMBIENT_TOKEN"],
|
||||
network_allowed=True,
|
||||
)
|
||||
assert granted.status is CheckStatus.PASSED
|
||||
assert "hunter2" in granted.output
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_run_command_blocks_egress_when_network_not_allowed(tmp_path: Path) -> None:
|
||||
command = CommandSpec(
|
||||
name="egress probe",
|
||||
argv=[
|
||||
sys.executable,
|
||||
"-c",
|
||||
(
|
||||
"import socket, sys; s = socket.socket(); s.settimeout(3); "
|
||||
"sys.exit(0 if s.connect_ex(('1.1.1.1', 53)) == 0 else 1)"
|
||||
),
|
||||
],
|
||||
)
|
||||
|
||||
result = await run_command(tmp_path, command)
|
||||
|
||||
if _network_isolation_prefix() is None:
|
||||
assert result.status is CheckStatus.UNAVAILABLE
|
||||
assert "not run" in result.output
|
||||
else:
|
||||
assert result.status is CheckStatus.FAILED
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_manifest_patch_includes_untracked_companion_file(tmp_path: Path) -> None:
|
||||
workspace, _ = _workspace(tmp_path)
|
||||
|
|
@ -432,18 +342,6 @@ async def test_manifest_patch_includes_untracked_companion_file(tmp_path: Path)
|
|||
_git(workspace, "diff", "--check")
|
||||
|
||||
|
||||
def test_skipped_check_and_missing_runtime_are_not_passing_evidence() -> None:
|
||||
assert command_status(0, "Skipping to avoid parser lock")[0] is CheckStatus.SKIPPED
|
||||
assert command_status(127, "bun: command not found")[0] is CheckStatus.UNAVAILABLE
|
||||
assert (
|
||||
not _regression()
|
||||
.model_copy(
|
||||
update={"base": _regression().base.model_copy(update={"failure_kind": "environment"})}
|
||||
)
|
||||
.passed()
|
||||
)
|
||||
|
||||
|
||||
def test_candidate_keeps_full_finding_without_inventing_reproduction() -> None:
|
||||
candidate = candidate_from_legacy_report(
|
||||
{
|
||||
|
|
@ -466,31 +364,6 @@ def test_candidate_keeps_full_finding_without_inventing_reproduction() -> None:
|
|||
assert candidate.reproduction is None
|
||||
|
||||
|
||||
def test_ambiguous_imports_and_syntax_never_authorize_environment_recovery() -> None:
|
||||
for output in [
|
||||
"Cannot find module '/tmp/test/skills/policy.js'",
|
||||
"No module named 'wrong_repo_path'",
|
||||
"SyntaxError: invalid syntax",
|
||||
]:
|
||||
status, kind = command_status(1, output)
|
||||
assert status is CheckStatus.FAILED
|
||||
assert kind == "unknown"
|
||||
assert command_status(127, "bun: command not found") == (CheckStatus.UNAVAILABLE, "environment")
|
||||
|
||||
|
||||
def test_regression_requires_consistent_execution_provenance() -> None:
|
||||
regression = _regression()
|
||||
for leg in (regression.base, regression.patched, regression.behavior):
|
||||
leg.source_digest = "a" * 64
|
||||
leg.environment_id = "execution:0"
|
||||
assert regression.passed()
|
||||
regression.base.environment_id = "execution:1"
|
||||
assert not regression.passed()
|
||||
regression.base.environment_id = "execution:0"
|
||||
regression.behavior = regression.behavior.model_copy(update={"source_digest": "b" * 64})
|
||||
assert not regression.passed()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_agent_tests_are_reused_without_controller_execution(tmp_path: Path) -> None:
|
||||
workspace, commit = _workspace(tmp_path)
|
||||
|
|
@ -506,9 +379,8 @@ async def test_agent_tests_are_reused_without_controller_execution(tmp_path: Pat
|
|||
)
|
||||
assert result.state is PreparationState.READY
|
||||
assert result.validation_mode == "agent_review"
|
||||
assert result.test_plan is None
|
||||
assert result.checks == outcome.command_results
|
||||
assert result.checks[0].tests_passed == 1
|
||||
assert "Ran 1 test" in result.checks[0].output
|
||||
assert result.prepared_source_digest == result.verifier.source_digest
|
||||
|
||||
|
||||
|
|
@ -534,7 +406,7 @@ async def test_review_can_request_more_than_two_repairs(tmp_path: Path) -> None:
|
|||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("defect", ["missing_unit", "failed", "empty", "stale", "missing_request"])
|
||||
@pytest.mark.parametrize("defect", ["missing_unit", "failed", "missing_request"])
|
||||
async def test_agent_approval_owns_test_evidence_without_controller_retries(
|
||||
tmp_path: Path, defect: str
|
||||
) -> None:
|
||||
|
|
@ -543,17 +415,14 @@ async def test_agent_approval_owns_test_evidence_without_controller_retries(
|
|||
async def repair(context, checks):
|
||||
result = await _noop_repair(context, checks)
|
||||
if defect == "missing_unit":
|
||||
result.command_results = [c for c in result.command_results if c.purpose != "unit"]
|
||||
result.command_results = [
|
||||
c for c in result.command_results if c.name != "existing suite"
|
||||
]
|
||||
elif defect == "missing_request":
|
||||
result.command_results = [c for c in result.command_results if c.purpose != "quality"]
|
||||
result.command_results = [c for c in result.command_results if c.name != "compile"]
|
||||
else:
|
||||
check = result.command_results[0]
|
||||
if defect == "failed":
|
||||
check.status, check.exit_code = CheckStatus.FAILED, 1
|
||||
elif defect == "empty":
|
||||
check.tests_passed = 0
|
||||
else:
|
||||
check.source_digest = "a" * 64
|
||||
check.status, check.exit_code = CheckStatus.FAILED, 1
|
||||
return result
|
||||
|
||||
request = _request(_candidate(commit))
|
||||
|
|
@ -581,12 +450,7 @@ async def test_incomplete_validation_can_be_finished_by_reviewer(tmp_path: Path)
|
|||
async def review(context, checks):
|
||||
assert checks[1].status is CheckStatus.FAILED
|
||||
command = CommandSpec(name="existing suite", argv=checks[1].argv, purpose="unit")
|
||||
executed = record_test_execution(
|
||||
await run_command(workspace, command, network_allowed=True), command
|
||||
)
|
||||
evidence[1] = executed.model_copy(
|
||||
update={"source_digest": evidence[0].source_digest, "environment_id": "test"}
|
||||
)
|
||||
evidence[1] = await _fixture_command(workspace, command)
|
||||
return await _verified(context, checks)
|
||||
|
||||
async def read_evidence():
|
||||
|
|
@ -631,12 +495,13 @@ async def test_interruptions_preserve_partial_patch_without_approval(
|
|||
raise AssertionError("Stopped repair must not be approved")
|
||||
|
||||
result = await prepare_fix(
|
||||
_request(_candidate(commit)),
|
||||
_request(_candidate(commit)).model_copy(
|
||||
update={"timeout_seconds": 1 if stop == "timeout" else 30}
|
||||
),
|
||||
workspace,
|
||||
repair=repair,
|
||||
verify=review,
|
||||
cancelled=lambda: cancel,
|
||||
policy=PreparationPolicy(timeout_seconds=1 if stop == "timeout" else 30),
|
||||
)
|
||||
assert result.state in {PreparationState.BLOCKED, PreparationState.FAILED}
|
||||
assert result.final_file_manifest
|
||||
|
|
@ -686,24 +551,6 @@ async def test_empty_deliverable_does_not_start_another_repair(tmp_path: Path) -
|
|||
assert "without a deliverable patch" in result.stop_reason
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("output", "expected"),
|
||||
[
|
||||
("Tests: 2 passed, 2 total", 2),
|
||||
("2 pass\n0 fail", 2),
|
||||
("# tests 2\n# pass 2\n# fail 0", 2),
|
||||
("Ran 3 tests in 0.1s\n\nOK (skipped=1)", 2),
|
||||
("3 examples, 0 failures, 1 pending", 2),
|
||||
("--- PASS: TestSafe (0.00s)", 1),
|
||||
("Ran 2 tests in 0.1s\n\nOK (skipped=2)", 0),
|
||||
("No tests found", None),
|
||||
],
|
||||
)
|
||||
def test_runner_summaries_record_actual_execution(output: str, expected: int | None) -> None:
|
||||
|
||||
assert passed_test_count(output) == expected
|
||||
|
||||
|
||||
def test_new_command_metadata_does_not_change_existing_finding_digest(tmp_path: Path) -> None:
|
||||
|
||||
_root, commit = _workspace(tmp_path)
|
||||
|
|
@ -715,21 +562,3 @@ def test_new_command_metadata_does_not_change_existing_finding_digest(tmp_path:
|
|||
json.dumps(payload, sort_keys=True, separators=(",", ":")).encode()
|
||||
).hexdigest()
|
||||
assert candidate.digest() == previous
|
||||
|
||||
|
||||
def test_unknown_runner_format_preserves_exit_result_for_independent_review() -> None:
|
||||
|
||||
command = CommandSpec(name="custom runner", argv=["./tests/run"], purpose="unit")
|
||||
result = record_test_execution(
|
||||
CheckResult(
|
||||
name=command.name,
|
||||
argv=command.argv,
|
||||
status=CheckStatus.PASSED,
|
||||
exit_code=0,
|
||||
duration_seconds=1,
|
||||
output="All project assertions completed successfully.",
|
||||
),
|
||||
command,
|
||||
)
|
||||
assert result.status is CheckStatus.PASSED
|
||||
assert result.tests_passed is None
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue