diff --git a/eval/tests/test_measure_evolution_cost.py b/eval/tests/test_measure_evolution_cost.py index db3094e57..b8b840b2f 100644 --- a/eval/tests/test_measure_evolution_cost.py +++ b/eval/tests/test_measure_evolution_cost.py @@ -43,18 +43,17 @@ def test_cells_are_submitted_run_major_arm_minor(): # At workers=3 that puts one cell of each arm in every wave. cells = task_cells(2, REVIEW_ARMS, 0) assert len(cells) == 6 - expected = [ - DURATIONS_BY_ARM[arm][run] + CELL_OVERHEAD_SECONDS - for run in range(2) - for arm in REVIEW_ARMS - ] + expected = [DURATIONS_BY_ARM[arm][run] for run in range(2) for arm in REVIEW_ARMS] assert cells == expected -def test_every_cell_carries_the_measured_overhead(): - assert task_cells(1, (CANDIDATE_ARM,), 0) == [ - DURATIONS_BY_ARM[CANDIDATE_ARM][0] + CELL_OVERHEAD_SECONDS - ] +def test_overhead_is_charged_serially_not_inside_the_pool(): + # The residual mixes per-cell work the pool divides with per-SHA setup it + # cannot, so it sits outside the schedule where more workers cannot + # dissolve it. + assert task_cells(1, (CANDIDATE_ARM,), 0) == [DURATIONS_BY_ARM[CANDIDATE_ARM][0]] + wide = generation_seconds(task_count=1, runs=3, arms=REVIEW_ARMS, workers=9, fed_pool=True) + assert wide >= PROPOSER_SECONDS + 9 * CELL_OVERHEAD_SECONDS # Cycling wraps, so a task can ask for more runs than the sample holds. long_sample = task_cells(len(DURATIONS_BY_ARM[CANDIDATE_ARM]) + 2, (CANDIDATE_ARM,), 0) assert len(long_sample) == len(DURATIONS_BY_ARM[CANDIDATE_ARM]) + 2 diff --git a/eval/workflow_bench/measure_evolution_cost.py b/eval/workflow_bench/measure_evolution_cost.py index 03bebeac2..cd0a9fb31 100644 --- a/eval/workflow_bench/measure_evolution_cost.py +++ b/eval/workflow_bench/measure_evolution_cost.py @@ -17,6 +17,7 @@ and only the candidate arm is paid. Cold assumes an empty seed. from __future__ import annotations import json +import math import re import statistics as st import subprocess @@ -45,7 +46,10 @@ DURATIONS_BY_ARM: dict[str, tuple[float, ...]] = { PROPOSER_SECONDS: float = MEASURED["proposer_duration_s"] _RESIDUAL = MEASURED["residual"] # Clone, graph build, sandbox, teardown: the sweep's own time, taken as that -# run's step wall minus what its sessions and proposer account for. +# run's step wall minus what its sessions and proposer account for. Charged +# SERIALLY, outside the pool. The residual mixes per-cell work the pool really +# does divide with per-SHA graph setup it cannot, and the artifact cannot +# separate them; serial is the pessimistic reading of an already-small term. CELL_OVERHEAD_SECONDS: float = _RESIDUAL["unaccounted_s"] / _RESIDUAL["cells"] # runner.py CANDIDATE_ARMS derives the candidate arm from its incumbent, and @@ -143,7 +147,7 @@ def task_cells(runs: int, arms: tuple[str, ...], offset: int) -> list[float]: for run_idx in range(runs): for arm in arms: sample = DURATIONS_BY_ARM[arm] - cells.append(sample[(offset + run_idx) % len(sample)] + CELL_OVERHEAD_SECONDS) + cells.append(sample[(offset + run_idx) % len(sample)]) return cells @@ -179,7 +183,9 @@ def expected_task_seconds( if runs < 1 or not arms: return 0.0 makespan = fed_makespan if fed_pool else wave_makespan - alignments = max(len(DURATIONS_BY_ARM[arm]) for arm in arms) + # lcm, not max: with samples of 13 and 14, max would wrap the shorter one + # and count its first entry twice. + alignments = math.lcm(*(len(DURATIONS_BY_ARM[arm]) for arm in arms)) return ( sum(makespan(task_cells(runs, arms, offset), workers) for offset in range(alignments)) / alignments @@ -189,11 +195,18 @@ def expected_task_seconds( def generation_seconds( *, task_count: int, runs: int, arms: tuple[str, ...], workers: int, fed_pool: bool ) -> int: - """Whole generation: one proposer session, then the tasks back to back.""" + """Whole generation: proposer, then the tasks back to back, plus overhead. + + Prices a HEALTHY sweep. A run whose cells return unusable evidence does not + reach this wall at all: the outage breaker aborts after + ``DEFAULT_OUTAGE_STREAK`` consecutive systemic failures, which for the + sample's own error sequence is cell 5 of 41. + """ return round( PROPOSER_SECONDS + task_count * expected_task_seconds(runs, arms, workers, fed_pool=fed_pool) + + task_count * runs * len(arms) * CELL_OVERHEAD_SECONDS ) diff --git a/eval/workflow_bench/session_durations.json b/eval/workflow_bench/session_durations.json index cd59cb90d..fe324a97d 100644 --- a/eval/workflow_bench/session_durations.json +++ b/eval/workflow_bench/session_durations.json @@ -61,6 +61,7 @@ "unaccounted_s": 2540.6, "cells": 41, "unique_shas": 5, - "_note": "Everything the sweep spent outside the agent sessions: per-SHA sanitize and `analyze --pdg --index-only`, plus each cell's clone, materialize, staging, sandbox and teardown. That run predates clone templates and graph prefetch, so this is an upper bound for the current code. The split between per-SHA and per-cell is not recoverable from the artifact, so the model charges it per cell and serially - the pessimistic reading of an already-small term." - } + "_note": "Everything the sweep spent outside the agent sessions: per-SHA sanitize and `analyze --pdg --index-only`, plus each cell's clone, materialize, staging, sandbox and teardown. That run predates clone templates and graph prefetch, so this is an upper bound for the current code. The split between per-SHA and per-cell is not recoverable from the artifact, so the model charges it per cell and serially, outside the pool - the pessimistic reading of an already-small term." + }, + "_breaker": "Replaying this sample's error_kind sequence through today's systemic_outage_streak trips the outage breaker at cell 5 of 41 (DEFAULT_OUTAGE_STREAK=5). The source run executed all 41, so its runner did not break on this sequence. The durations stay valid as per-cell timings; what they cannot describe is a 54-cell sweep with this failure profile, because the current code would never run one." }