From d5c396cc6e4f4ac96eb396dab0eba6d300c1c29c Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Sun, 6 Sep 2026 10:39:38 +0000 Subject: [PATCH] perf(eval): price arms separately and stop inventing setup constants MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two errors in the model, both found by auditing it against the artifact it claims to describe. The arms are not interchangeable. `candidate_review` runs 1416s at the mean against `review`'s 1204s and `ce_review`'s 1176s, and the weekly lane pays the candidate arm and nothing else — reuse skips both incumbents. Pricing weekly from a pooled sample charged it for arms it never runs: weekly is 4.59h, not the 3.65h a pooled sample reported. Cells are also submitted run-major and arm-minor, so at workers=3 every wave holds one cell of each arm and the slowest arm sets the wave; the model now builds cells in that order. The setup constants were invented. GRAPH_ANALYZE_SECONDS=600 and TEMPLATE_SANITIZE_SECONDS=180 charged 3900s of per-SHA setup for a cold run — more than the entire non-session time of the source run, which was 2541s for 41 cells and 5 SHAs. `duration_s` is the sum of a cell's Claude sessions (runner_sessions.py), so that 2541s residual is every clone, graph build, sandbox and teardown the sweep paid. The model now charges the measured residual per cell, 62.0s, and no longer credits clone templates or graph prefetch: both landed after that run and there is no measurement of them yet. The residual bounds what they can be worth. Cold 37452s (10.40h), weekly 16541s (4.59h), against a fed pool at 31683s and 16541s. Measurement only; no runtime behaviour changes. Co-Authored-By: Claude Opus 5 (1M context) Co-Authored-By: Claude Opus 5 (1M context) --- eval/tests/test_measure_evolution_cost.py | 97 ++--- eval/workflow_bench/measure_evolution_cost.py | 366 +++++------------- eval/workflow_bench/session_durations.json | 108 +++--- 3 files changed, 221 insertions(+), 350 deletions(-) diff --git a/eval/tests/test_measure_evolution_cost.py b/eval/tests/test_measure_evolution_cost.py index 21e7445b3..db3094e57 100644 --- a/eval/tests/test_measure_evolution_cost.py +++ b/eval/tests/test_measure_evolution_cost.py @@ -5,49 +5,59 @@ from __future__ import annotations import pytest from workflow_bench.measure_evolution_cost import ( - CELL_COPY_SECONDS, - CELL_DURATIONS, - GRAPH_ANALYZE_SECONDS, + CANDIDATE_ARM, + CELL_OVERHEAD_SECONDS, + DURATIONS_BY_ARM, PROPOSER_SECONDS, - TEMPLATE_SANITIZE_SECONDS, - cell_durations, - expected_makespan, + REVIEW_ARMS, + expected_task_seconds, fed_makespan, fed_pool_enabled, + generation_seconds, graph_pipeline_enabled, - paid_cells_per_task, - setup_wall_seconds, - unique_paid_shas, + paid_arms, + task_cells, wave_makespan, ) -TASKS = [ - {"id": "a", "ref": "sha-1"}, - {"id": "b", "ref": "sha-2"}, - {"id": "c", "ref": "sha-1"}, -] - -def test_measured_sample_is_present_and_unsorted(): - # Sorting would hand each task a uniform block and hide the variance the - # whole model exists to price. - assert len(CELL_DURATIONS) >= 20 - assert list(CELL_DURATIONS) != sorted(CELL_DURATIONS) +def test_every_arm_has_its_own_unsorted_sample(): + assert set(DURATIONS_BY_ARM) == set(REVIEW_ARMS) + for arm, sample in DURATIONS_BY_ARM.items(): + assert len(sample) >= 10, arm + # Sorting would hand each task a uniform block and hide the variance + # the whole model exists to price. + assert list(sample) != sorted(sample), arm assert PROPOSER_SECONDS > 0 + assert CELL_OVERHEAD_SECONDS > 0 -def test_weekly_reuse_pays_only_candidate_cells(): - assert paid_cells_per_task(TASKS, runs=3, weekly=True, reuse_enabled=True) == [3, 3, 3] - assert paid_cells_per_task(TASKS, runs=3, weekly=False, reuse_enabled=True) == [9, 9, 9] - assert paid_cells_per_task(TASKS, runs=3, weekly=True, reuse_enabled=False) == [9, 9, 9] +def test_weekly_reuse_pays_the_candidate_arm_only(): + assert paid_arms(weekly=True, reuse_enabled=True) == (CANDIDATE_ARM,) + assert paid_arms(weekly=False, reuse_enabled=True) == REVIEW_ARMS + assert paid_arms(weekly=True, reuse_enabled=False) == REVIEW_ARMS -def test_every_cell_carries_its_own_clone(): - durations = cell_durations(3, True) - assert durations == [d + CELL_COPY_SECONDS for d in CELL_DURATIONS[:3]] - assert cell_durations(2, False) == [d + TEMPLATE_SANITIZE_SECONDS for d in CELL_DURATIONS[:2]] - # Cycling wraps, so a task can be longer than the sample. - assert len(cell_durations(len(CELL_DURATIONS) + 5, True)) == len(CELL_DURATIONS) + 5 +def test_cells_are_submitted_run_major_arm_minor(): + # runner.py: [(run_idx, arm) for run_idx in range(runs) for arm in arms]. + # At workers=3 that puts one cell of each arm in every wave. + cells = task_cells(2, REVIEW_ARMS, 0) + assert len(cells) == 6 + expected = [ + DURATIONS_BY_ARM[arm][run] + CELL_OVERHEAD_SECONDS + for run in range(2) + for arm in REVIEW_ARMS + ] + assert cells == expected + + +def test_every_cell_carries_the_measured_overhead(): + assert task_cells(1, (CANDIDATE_ARM,), 0) == [ + DURATIONS_BY_ARM[CANDIDATE_ARM][0] + CELL_OVERHEAD_SECONDS + ] + # Cycling wraps, so a task can ask for more runs than the sample holds. + long_sample = task_cells(len(DURATIONS_BY_ARM[CANDIDATE_ARM]) + 2, (CANDIDATE_ARM,), 0) + assert len(long_sample) == len(DURATIONS_BY_ARM[CANDIDATE_ARM]) + 2 def test_a_wave_costs_its_slowest_cell_and_a_fed_pool_does_not(): @@ -59,21 +69,20 @@ def test_a_wave_costs_its_slowest_cell_and_a_fed_pool_does_not(): assert fed_makespan(slow, 1) == wave_makespan(slow, 1) == 24.0 -def test_expected_makespan_is_rotation_averaged_and_deterministic(): - first = expected_makespan(9, 3, fed_pool=False, clone_templates_enabled=True) - assert first == expected_makespan(9, 3, fed_pool=False, clone_templates_enabled=True) - assert expected_makespan(0, 3, fed_pool=False, clone_templates_enabled=True) == 0.0 +def test_expected_task_seconds_is_alignment_averaged_and_deterministic(): + waved = expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=False) + assert waved == expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=False) + assert expected_task_seconds(0, REVIEW_ARMS, 3, fed_pool=False) == 0.0 + assert expected_task_seconds(3, (), 3, fed_pool=False) == 0.0 # The barrier can only cost time, never save it. - assert first >= expected_makespan(9, 3, fed_pool=True, clone_templates_enabled=True) + assert waved >= expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=True) -def test_setup_counts_each_sha_once_and_no_longer_charges_cells(): - assert unique_paid_shas(TASKS, [9, 9, 9]) == 2 - assert unique_paid_shas(TASKS, [9, 0, 0]) == 1 - assert setup_wall_seconds( - unique_shas=2, paid_per_task=[9, 9, 9], clone_templates_enabled=True - ) == 2 * (TEMPLATE_SANITIZE_SECONDS + GRAPH_ANALYZE_SECONDS) - assert setup_wall_seconds(unique_shas=2, paid_per_task=[0], clone_templates_enabled=True) == 0 +def test_a_generation_pays_one_proposer_session_on_top_of_its_tasks(): + one = generation_seconds(task_count=1, runs=3, arms=REVIEW_ARMS, workers=3, fed_pool=False) + two = generation_seconds(task_count=2, runs=3, arms=REVIEW_ARMS, workers=3, fed_pool=False) + # Each extra task adds exactly one task's makespan; the proposer is paid once. + assert two - one == pytest.approx(one - PROPOSER_SECONDS, abs=1.0) def test_feature_flags_read_the_runner_not_the_wish(): @@ -85,5 +94,5 @@ def test_feature_flags_read_the_runner_not_the_wish(): @pytest.mark.parametrize("workers", [1, 3, 8]) def test_more_workers_never_lengthen_a_task(workers): - serial = expected_makespan(9, 1, fed_pool=True, clone_templates_enabled=True) - assert expected_makespan(9, workers, fed_pool=True, clone_templates_enabled=True) <= serial + serial = expected_task_seconds(3, REVIEW_ARMS, 1, fed_pool=True) + assert expected_task_seconds(3, REVIEW_ARMS, workers, fed_pool=True) <= serial diff --git a/eval/workflow_bench/measure_evolution_cost.py b/eval/workflow_bench/measure_evolution_cost.py index 5448bdc58..03bebeac2 100644 --- a/eval/workflow_bench/measure_evolution_cost.py +++ b/eval/workflow_bench/measure_evolution_cost.py @@ -1,47 +1,28 @@ #!/usr/bin/env python3 """Cheap cost model for the skill-evolution review generation. -This is the ce-optimize measurement harness. It does not start Claude or -replay Actions run 33962002890. Wall clock is session waves plus the -pre-sweep setup the runner still pays before ``sweep_task_cells``: -``make_worktree`` + ``sanitize_clone_for_hidden_oracles`` and -``analyze --pdg --index-only``. +This is the ce-optimize measurement harness. It does not start Claude and it +does not replay a run. It reads the review corpus, the evolve defaults and the +workflow's workers default, then schedules the measured cell durations in +``session_durations.json`` the way ``sweep_task_cells`` schedules real cells. -Weekly assumes a matching seed so every reusable comparator cell is skipped. -Cold assumes an empty seed. Candidate cells are never treated as reusable. +Everything priced here is measured. Cell durations and the proposer session +come from a real artifact, and the work outside the agent sessions comes from +that run's own step wall minus the time its sessions and proposer account for. + +Weekly assumes a matching seed, so every reusable comparator cell is skipped +and only the candidate arm is paid. Cold assumes an empty seed. """ from __future__ import annotations import json -import math import re import statistics as st import subprocess import sys from pathlib import Path -# Measured cell durations, not a mean: a wave waits for its SLOWEST cell, and -# these run 826s at the median against a 5400s session ceiling, so a model -# built on an average understates every concurrent schedule. See -# session_durations.json for provenance and its sampling caveat. -DURATIONS = json.loads( - (Path(__file__).resolve().parent / "session_durations.json").read_text(encoding="utf-8") -) -CELL_DURATIONS = tuple(DURATIONS["cell_duration_s"]) -# One proposer session per generation, ahead of the benchmark and unavoidably -# on the critical path. -PROPOSER_SECONDS = DURATIONS["proposer_duration_s"] -# Conservative mean for `analyze --pdg --index-only` of a sanitized -# GitNexus snapshot on the evolution box. The hard timeout is 3600s. -GRAPH_ANALYZE_SECONDS = 600 -# `git clone --no-local` plus sanitize_clone_for_hidden_oracles (repack / -# prune / fsck). README called this "minutes" per cell before templates. -TEMPLATE_SANITIZE_SECONDS = 180 -# copy_isolated_tree of an already-sanitized template (reflink or copy). -# run_cell does this inside its pool worker, so it is a per-wave cost. -CELL_COPY_SECONDS = 15 - REPO_ROOT = Path(__file__).resolve().parents[2] EVAL_ROOT = REPO_ROOT / "eval" REVIEW_TASKS = EVAL_ROOT / "workflow_bench" / "tasks.review.scenarios.yaml" @@ -51,18 +32,35 @@ ARTIFACTS_PY = EVAL_ROOT / "workflow_bench" / "runner_artifacts.py" REUSE_PY = EVAL_ROOT / "workflow_bench" / "comparator_reuse.py" WORKFLOW = REPO_ROOT / ".github" / "workflows" / "gitnexus-skill-evolution.yml" -REVIEW_ARMS = ("ce_review", "review", "candidate_review") +MEASURED = json.loads( + (Path(__file__).resolve().parent / "session_durations.json").read_text(encoding="utf-8") +) +# Per arm, because the arms are not interchangeable and the weekly lane pays +# only the candidate one. Cells are submitted run-major and arm-minor +# (runner.py ``planned``), so at workers=3 every wave holds one cell of each +# arm and the slowest arm sets the wave. +DURATIONS_BY_ARM: dict[str, tuple[float, ...]] = { + arm: tuple(values) for arm, values in MEASURED["cell_duration_s_by_arm"].items() +} +PROPOSER_SECONDS: float = MEASURED["proposer_duration_s"] +_RESIDUAL = MEASURED["residual"] +# Clone, graph build, sandbox, teardown: the sweep's own time, taken as that +# run's step wall minus what its sessions and proposer account for. +CELL_OVERHEAD_SECONDS: float = _RESIDUAL["unaccounted_s"] / _RESIDUAL["cells"] + +# runner.py CANDIDATE_ARMS derives the candidate arm from its incumbent, and +# only an incumbent row can be reused from a prior generation. CANDIDATE_ARM = "candidate_review" -REUSABLE_ARMS = frozenset({"review", "ce_review"}) +REVIEW_ARMS = ("ce_review", "review", CANDIDATE_ARM) SUITE_FILES = ( + "tests/test_measure_evolution_cost.py", "tests/test_comparator_reuse.py", "tests/test_evolve.py", "tests/test_sanitized_graph.py", + "tests/test_workflow_bench.py", "tests/test_workflow_bench_sessions.py", "tests/test_session_progress.py", - "tests/test_measure_evolution_cost.py", - "tests/test_workflow_bench.py", ) @@ -87,22 +85,14 @@ def review_tasks(text: str) -> list[dict[str, str]]: def evolve_default(name: str, text: str) -> int: - match = re.search( - rf'add_argument\("--{re.escape(name)}".*?default=(\d+)', - text, - flags=re.S, - ) + match = re.search(rf'add_argument\("--{re.escape(name)}".*?default=(\d+)', text, flags=re.S) if match is None: raise ValueError(f"evolve.py is missing --{name} default") return int(match.group(1)) def workflow_dispatch_workers(text: str) -> int: - match = re.search( - r"^\s+workers:\n(?:.*\n)*?^\s+default: '(\d+)'", - text, - flags=re.M, - ) + match = re.search(r"^\s+workers:\n(?:.*\n)*?^\s+default: '(\d+)'", text, flags=re.M) if match is None: raise ValueError("workflow_dispatch workers default is missing") return int(match.group(1)) @@ -118,159 +108,47 @@ def feature_enabled() -> tuple[int, int]: and "select_reusable_comparator_rows" in runner and "CANDIDATE" in _read(REUSE_PY) ) - templates = int( - "def copy_isolated_tree" in artifacts - and "clone_templates" in runner - and "clone_template" in runner - ) + templates = int("def copy_isolated_tree" in artifacts and "clone_templates" in runner) return reuse, templates -def paid_cells_per_task( - tasks: list[dict[str, str]], - *, - runs: int, - weekly: bool, - reuse_enabled: bool, -) -> list[int]: - per_task: list[int] = [] - for _task in tasks: - paid = 0 - for _run in range(runs): - for arm in REVIEW_ARMS: - if weekly and reuse_enabled and arm in REUSABLE_ARMS: - continue - paid += 1 - per_task.append(paid) - return per_task - - -def unique_paid_shas(tasks: list[dict[str, str]], paid_per_task: list[int]) -> int: - seen: set[str] = set() - for task, paid in zip(tasks, paid_per_task, strict=True): - ref = task.get("ref", "") - if paid > 0 and ref: - seen.add(ref) - return len(seen) - - -def sha_setup_seconds(clone_templates_enabled: bool) -> int: - if clone_templates_enabled: - return TEMPLATE_SANITIZE_SECONDS + GRAPH_ANALYZE_SECONDS - return GRAPH_ANALYZE_SECONDS - - -def setup_wall_seconds( - *, - unique_shas: int, - paid_per_task: list[int], - clone_templates_enabled: bool, -) -> int: - """Serial per-SHA graph setup. Per-cell clones ride inside the cell.""" - - if unique_shas < 1 or sum(paid_per_task) < 1: - return 0 - return unique_shas * sha_setup_seconds(clone_templates_enabled) - - def graph_pipeline_enabled(runner_text: str) -> int: - """True only when the runner prefetches the next SHA during paid sessions.""" + """True when the runner prefetches the next SHA during paid sessions.""" - return int( - bool( - re.search( - r"(graph_prefetch|prefetch_next_graph|pipeline_next_sha|prefetch_clone_template)", - runner_text, - ) - ) - ) + return int("prefetch_next_graph" in runner_text or "GraphPrefetch" in runner_text) -def pipelined_wall_seconds( - tasks: list[dict[str, str]], - paid_per_task: list[int], - session_per_task: list[float], - *, - clone_templates_enabled: bool, -) -> int: - """Overlap the next unseen SHA's setup with the current task's session wave. +def fed_pool_enabled(runner_text: str) -> int: + """True when the sweep feeds a live pool instead of waiting on waves.""" - The first SHA still pays setup up front. Later SHAs hide behind the - previous task's paid sessions when that wave is longer than SHA setup. - Per-cell copies stay on the task's critical path. + return int("def _run_fed_pool" in runner_text) + + +def paid_arms(weekly: bool, reuse_enabled: bool) -> tuple[str, ...]: + """Arms a generation actually pays for.""" + + if weekly and reuse_enabled: + return (CANDIDATE_ARM,) + return REVIEW_ARMS + + +def task_cells(runs: int, arms: tuple[str, ...], offset: int) -> list[float]: + """One task's cell durations in submission order: run-major, arm-minor. + + Each arm draws from its own measured sample, cycled from ``offset`` so the + caller can average over every alignment instead of trusting one. """ - setup = sha_setup_seconds(clone_templates_enabled) - elapsed = PROPOSER_SECONDS - ready: set[str] = set() - in_flight: tuple[str, int] | None = None - for index, (task, paid, session) in enumerate(zip(tasks, paid_per_task, session_per_task, strict=True)): - if paid <= 0: - continue - sha = task.get("ref", "") - if sha not in ready: - if in_flight is not None and in_flight[0] == sha: - elapsed = max(elapsed, in_flight[1]) - ready.add(sha) - in_flight = None - else: - elapsed += setup - if sha: - ready.add(sha) - session_start = elapsed - elapsed += session - if in_flight is None: - for later_task, later_paid in zip(tasks[index + 1 :], paid_per_task[index + 1 :], strict=True): - later_sha = later_task.get("ref", "") - if later_paid > 0 and later_sha and later_sha not in ready: - in_flight = (later_sha, session_start + setup) - break - elif in_flight[1] <= elapsed: - ready.add(in_flight[0]) - in_flight = None - return round(elapsed) - - -def cell_durations(count: int, clone_templates_enabled: bool, offset: int = 0) -> list[float]: - """Deterministic per-cell durations: the measured sample, cycled from ``offset``. - - Cycled rather than sampled so every scheduler is scored against the same - cells and the harness stays reproducible. Each cell carries its own clone, - which ``run_cell`` pays inside its worker. - """ - - clone = CELL_COPY_SECONDS if clone_templates_enabled else TEMPLATE_SANITIZE_SECONDS - size = len(CELL_DURATIONS) - return [CELL_DURATIONS[(offset + i) % size] + clone for i in range(count)] - - -def expected_makespan( - count: int, - workers: int, - *, - fed_pool: bool, - clone_templates_enabled: bool, -) -> float: - """Mean makespan over every rotation of the measured sample. - - One fixed alignment would let an accident of the source run — its slowest - cells happen to come first — decide the answer. Averaging all rotations - keeps the real multiset and the real ordering effects while removing that - alignment artifact, and stays deterministic. - """ - - if count <= 0: - return 0.0 - makespan = fed_makespan if fed_pool else wave_makespan - totals = [ - makespan(cell_durations(count, clone_templates_enabled, offset), workers) - for offset in range(len(CELL_DURATIONS)) - ] - return sum(totals) / len(totals) + cells: list[float] = [] + for run_idx in range(runs): + for arm in arms: + sample = DURATIONS_BY_ARM[arm] + cells.append(sample[(offset + run_idx) % len(sample)] + CELL_OVERHEAD_SECONDS) + return cells def wave_makespan(durations: list[float], workers: int) -> float: - """Current scheduler: fixed waves of ``workers``, barrier between them.""" + """Today's scheduler: fixed waves of ``workers``, with a barrier between.""" return sum( max(durations[start : start + workers]) for start in range(0, len(durations), workers) @@ -287,28 +165,36 @@ def fed_makespan(durations: list[float], workers: int) -> float: return max(busy_until) -def session_wall_seconds( - paid_per_task: list[int], - workers: int, - *, - fed_pool: bool, - clone_templates_enabled: bool, -) -> int: - if workers < 1: - raise ValueError("workers must be at least 1") +def expected_task_seconds( + runs: int, arms: tuple[str, ...], workers: int, *, fed_pool: bool +) -> float: + """Mean makespan of one task over every alignment of the measured samples. + + One fixed alignment would let an accident of the source run - its slowest + cells happen to come first - decide the answer. Averaging keeps the real + multiset and the real ordering effects without that artifact, and stays + deterministic. + """ + + if runs < 1 or not arms: + return 0.0 makespan = fed_makespan if fed_pool else wave_makespan - total = 0.0 - for paid in paid_per_task: - if paid <= 0: - continue - total += makespan(cell_durations(paid, clone_templates_enabled), workers) - return round(total) + alignments = max(len(DURATIONS_BY_ARM[arm]) for arm in arms) + return ( + sum(makespan(task_cells(runs, arms, offset), workers) for offset in range(alignments)) + / alignments + ) -def fed_pool_enabled(runner_text: str) -> int: - """True only when the sweep feeds a live pool instead of waiting on waves.""" +def generation_seconds( + *, task_count: int, runs: int, arms: tuple[str, ...], workers: int, fed_pool: bool +) -> int: + """Whole generation: one proposer session, then the tasks back to back.""" - return int("def _run_fed_pool" in runner_text) + return round( + PROPOSER_SECONDS + + task_count * expected_task_seconds(runs, arms, workers, fed_pool=fed_pool) + ) def _pytest_python() -> list[str]: @@ -324,23 +210,10 @@ def suite_passed() -> int: files = [name for name in SUITE_FILES if (EVAL_ROOT / name).is_file()] if not files: return 0 - cmd = [ - *_pytest_python(), - "-m", - "pytest", - *files, - "-q", - "--tb=no", - "--no-header", - ] + cmd = [*_pytest_python(), "-m", "pytest", *files, "-q", "--tb=no", "--no-header"] try: completed = subprocess.run( - cmd, - cwd=EVAL_ROOT, - check=False, - capture_output=True, - text=True, - timeout=240, + cmd, cwd=EVAL_ROOT, check=False, capture_output=True, text=True, timeout=240 ) except (OSError, subprocess.TimeoutExpired): return 0 @@ -352,70 +225,43 @@ def main() -> int: evolve = _read(EVOLVE_PY) runner = _read(RUNNER_PY) runs = evolve_default("runs", evolve) - promotion_min_runs = evolve_default("promotion-min-runs", evolve) workers = workflow_dispatch_workers(_read(WORKFLOW)) reuse_enabled, clone_templates_enabled = feature_enabled() - pipeline_enabled = graph_pipeline_enabled(runner) fed_pool = fed_pool_enabled(runner) - templates = bool(clone_templates_enabled) payload: dict[str, object] = {} for label, weekly in (("weekly", True), ("cold", False)): - paid = paid_cells_per_task( - tasks, runs=runs, weekly=weekly, reuse_enabled=bool(reuse_enabled) + arms = paid_arms(weekly, bool(reuse_enabled)) + payload[f"estimated_{label}_wall_seconds"] = generation_seconds( + task_count=len(tasks), runs=runs, arms=arms, workers=workers, fed_pool=bool(fed_pool) ) - per_task = [ - expected_makespan( - count, workers, fed_pool=bool(fed_pool), clone_templates_enabled=templates - ) - for count in paid - ] - sessions = round(sum(per_task)) - setup = setup_wall_seconds( - unique_shas=unique_paid_shas(tasks, paid), - paid_per_task=paid, - clone_templates_enabled=templates, - ) - if pipeline_enabled: - wall = pipelined_wall_seconds( - tasks, paid, per_task, clone_templates_enabled=templates - ) - setup = wall - sessions - else: - wall = round(sessions + setup + PROPOSER_SECONDS) - payload[f"estimated_{label}_wall_seconds"] = wall - payload[f"paid_{label}_cells"] = sum(paid) - payload[f"session_{label}_seconds"] = sessions - payload[f"setup_{label}_seconds"] = setup - # What the barrier costs: the same cells under a continuously fed pool. - payload[f"fed_pool_{label}_session_seconds"] = round( - sum( - expected_makespan( - count, workers, fed_pool=True, clone_templates_enabled=templates - ) - for count in paid - ) + payload[f"paid_{label}_cells"] = len(tasks) * runs * len(arms) + # What the wave barrier costs: the same cells, continuously fed. + payload[f"fed_pool_{label}_wall_seconds"] = generation_seconds( + task_count=len(tasks), runs=runs, arms=arms, workers=workers, fed_pool=True ) + all_durations = [d for sample in DURATIONS_BY_ARM.values() for d in sample] payload.update( { "suite_passed": suite_passed(), - "promotion_min_runs": promotion_min_runs, + "promotion_min_runs": evolve_default("promotion-min-runs", evolve), "review_task_count": len(tasks), "candidate_cells": len(tasks) * runs, "workers": workers, "unique_task_shas": len({t.get("ref", "") for t in tasks if t.get("ref")}), "reuse_enabled": reuse_enabled, "clone_templates_enabled": clone_templates_enabled, - "graph_pipeline_enabled": pipeline_enabled, + "graph_pipeline_enabled": graph_pipeline_enabled(runner), "fed_pool_enabled": fed_pool, - "measured_cell_count": len(CELL_DURATIONS), - "median_cell_seconds": round(st.median(CELL_DURATIONS)), - "mean_cell_seconds": round(st.mean(CELL_DURATIONS)), - "max_cell_seconds": round(max(CELL_DURATIONS)), + "measured_cell_count": len(all_durations), + "median_cell_seconds": round(st.median(all_durations)), + "mean_cell_seconds": round(st.mean(all_durations)), + "max_cell_seconds": round(max(all_durations)), + "median_candidate_cell_seconds": round(st.median(DURATIONS_BY_ARM[CANDIDATE_ARM])), + "mean_candidate_cell_seconds": round(st.mean(DURATIONS_BY_ARM[CANDIDATE_ARM])), "proposer_seconds": round(PROPOSER_SECONDS), - "graph_analyze_seconds": GRAPH_ANALYZE_SECONDS, - "template_sanitize_seconds": TEMPLATE_SANITIZE_SECONDS, + "cell_overhead_seconds": round(CELL_OVERHEAD_SECONDS, 1), } ) json.dump(payload, sys.stdout, sort_keys=True) diff --git a/eval/workflow_bench/session_durations.json b/eval/workflow_bench/session_durations.json index 838643655..cd59cb90d 100644 --- a/eval/workflow_bench/session_durations.json +++ b/eval/workflow_bench/session_durations.json @@ -1,50 +1,66 @@ { - "_provenance": "Actions run 33912693948 (2026-09-04), review profile, workers=1, gen-0. Artifact gitnexus-evolution-33912693948-1, gen-0/bench/results.jsonl.", - "_caveat": "Every cell in that run was unusable evidence (32 review-evidence-invalid, 6 session-error, 3 skill-not-invoked) and two hit the 5400s session ceiling. Durations are real; a run that resolves cleanly may sit lower.", + "_provenance": "Actions run 33912693948 (2026-09-04), review profile, gen-0, workers=1. Artifact gitnexus-evolution-33912693948-1: gen-0/bench/results.jsonl and gen-0/proposer-session.json. Step wall from the Actions API.", + "_caveat": "Every cell in that run returned unusable evidence (32 review-evidence-invalid, 6 session-error, 3 skill-not-invoked); two hit the 5400s ceiling and it cost 653. Durations are real, but a run that resolves cleanly may sit lower. It is the only live artifact - the 2026-07-22 green run's has expired.", + "_order": "Submission order, deliberately unsorted: the model cycles these, so sorting would hand each task a uniform block and hide the variance being measured.", + "_duration_scope": "duration_s is the sum of the cell's Claude session durations (runner_sessions.py). It excludes the clone, graph materialize, asset staging, sandbox setup and teardown - those live in the residual below.", "session_ceiling_s": 5400, "proposer_duration_s": 344.7, - "cell_duration_s": [ - 3744.6, - 5400.0, - 2485.6, - 2140.4, - 2976.4, - 1338.0, - 1418.4, - 1162.8, - 3075.2, - 436.1, - 627.1, - 702.5, - 653.3, - 901.3, - 1240.4, - 1022.3, - 436.9, - 5400.0, - 851.2, - 704.3, - 762.1, - 1191.1, - 963.4, - 342.9, - 902.1, - 631.8, - 826.3, - 847.2, - 627.9, - 489.7, - 502.6, - 663.5, - 675.4, - 991.3, - 662.4, - 337.8, - 1222.7, - 361.3, - 734.3, - 544.0, - 741.1 - ], - "_order": "Submission order from the source run, deliberately unsorted: the model cycles this list across cells, so sorting it would hand every task a uniform block and hide exactly the variance being measured." + "cell_duration_s_by_arm": { + "candidate_review": [ + 2485.6, + 1338.0, + 3075.2, + 702.5, + 1240.4, + 5400.0, + 762.1, + 342.9, + 826.3, + 489.7, + 675.4, + 337.8, + 734.3 + ], + "ce_review": [ + 3744.6, + 2140.4, + 1418.4, + 436.1, + 653.3, + 1022.3, + 851.2, + 1191.1, + 902.1, + 847.2, + 502.6, + 991.3, + 1222.7, + 544.0 + ], + "review": [ + 5400.0, + 2976.4, + 1162.8, + 627.1, + 901.3, + 436.9, + 704.3, + 963.4, + 631.8, + 627.9, + 663.5, + 662.4, + 361.3, + 741.1 + ] + }, + "residual": { + "benchmark_step_wall_s": 54623, + "session_seconds": 51737.7, + "proposer_seconds": 344.7, + "unaccounted_s": 2540.6, + "cells": 41, + "unique_shas": 5, + "_note": "Everything the sweep spent outside the agent sessions: per-SHA sanitize and `analyze --pdg --index-only`, plus each cell's clone, materialize, staging, sandbox and teardown. That run predates clone templates and graph prefetch, so this is an upper bound for the current code. The split between per-SHA and per-cell is not recoverable from the artifact, so the model charges it per cell and serially - the pessimistic reading of an already-small term." + } }