mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-09 03:17:54 +00:00
perf(eval): charge sweep overhead where more workers cannot dissolve it
Three defects, found by auditing the model against the artifact again. The overhead was charged inside the schedule. session_durations.json claimed the residual was charged "per cell and serially - the pessimistic reading", but task_cells folded it into each cell's duration, where the pool then divided it by the worker count. The residual mixes per-cell work the pool really does divide with per-SHA graph setup it cannot, and the artifact cannot separate them, so it now sits outside the schedule: cold 11.09h, not 10.40h. Alignment averaging weighted the shortest sample twice. The arm samples are 13, 14 and 14 long and the average ran over max()=14 offsets, so candidate_review's first cell was counted twice and its last never. Averaging over lcm()=182 offsets weights every arm's sample evenly. The wall assumed all 54 cells run. Replaying the sample's own error_kind sequence through today's systemic_outage_streak trips the outage breaker at cell 5 of 41. The source run executed all 41, so its runner did not break on that sequence, but the current one would: these numbers price a HEALTHY sweep, and a sweep with the sample's failure profile never reaches them. Stated on generation_seconds and recorded next to the sample it qualifies. Measurement only; no runtime behaviour changes. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
d5c396cc6e
commit
87a5764c5f
3 changed files with 28 additions and 15 deletions
|
|
@ -43,18 +43,17 @@ def test_cells_are_submitted_run_major_arm_minor():
|
|||
# At workers=3 that puts one cell of each arm in every wave.
|
||||
cells = task_cells(2, REVIEW_ARMS, 0)
|
||||
assert len(cells) == 6
|
||||
expected = [
|
||||
DURATIONS_BY_ARM[arm][run] + CELL_OVERHEAD_SECONDS
|
||||
for run in range(2)
|
||||
for arm in REVIEW_ARMS
|
||||
]
|
||||
expected = [DURATIONS_BY_ARM[arm][run] for run in range(2) for arm in REVIEW_ARMS]
|
||||
assert cells == expected
|
||||
|
||||
|
||||
def test_every_cell_carries_the_measured_overhead():
|
||||
assert task_cells(1, (CANDIDATE_ARM,), 0) == [
|
||||
DURATIONS_BY_ARM[CANDIDATE_ARM][0] + CELL_OVERHEAD_SECONDS
|
||||
]
|
||||
def test_overhead_is_charged_serially_not_inside_the_pool():
|
||||
# The residual mixes per-cell work the pool divides with per-SHA setup it
|
||||
# cannot, so it sits outside the schedule where more workers cannot
|
||||
# dissolve it.
|
||||
assert task_cells(1, (CANDIDATE_ARM,), 0) == [DURATIONS_BY_ARM[CANDIDATE_ARM][0]]
|
||||
wide = generation_seconds(task_count=1, runs=3, arms=REVIEW_ARMS, workers=9, fed_pool=True)
|
||||
assert wide >= PROPOSER_SECONDS + 9 * CELL_OVERHEAD_SECONDS
|
||||
# Cycling wraps, so a task can ask for more runs than the sample holds.
|
||||
long_sample = task_cells(len(DURATIONS_BY_ARM[CANDIDATE_ARM]) + 2, (CANDIDATE_ARM,), 0)
|
||||
assert len(long_sample) == len(DURATIONS_BY_ARM[CANDIDATE_ARM]) + 2
|
||||
|
|
|
|||
|
|
@ -17,6 +17,7 @@ and only the candidate arm is paid. Cold assumes an empty seed.
|
|||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
import re
|
||||
import statistics as st
|
||||
import subprocess
|
||||
|
|
@ -45,7 +46,10 @@ DURATIONS_BY_ARM: dict[str, tuple[float, ...]] = {
|
|||
PROPOSER_SECONDS: float = MEASURED["proposer_duration_s"]
|
||||
_RESIDUAL = MEASURED["residual"]
|
||||
# Clone, graph build, sandbox, teardown: the sweep's own time, taken as that
|
||||
# run's step wall minus what its sessions and proposer account for.
|
||||
# run's step wall minus what its sessions and proposer account for. Charged
|
||||
# SERIALLY, outside the pool. The residual mixes per-cell work the pool really
|
||||
# does divide with per-SHA graph setup it cannot, and the artifact cannot
|
||||
# separate them; serial is the pessimistic reading of an already-small term.
|
||||
CELL_OVERHEAD_SECONDS: float = _RESIDUAL["unaccounted_s"] / _RESIDUAL["cells"]
|
||||
|
||||
# runner.py CANDIDATE_ARMS derives the candidate arm from its incumbent, and
|
||||
|
|
@ -143,7 +147,7 @@ def task_cells(runs: int, arms: tuple[str, ...], offset: int) -> list[float]:
|
|||
for run_idx in range(runs):
|
||||
for arm in arms:
|
||||
sample = DURATIONS_BY_ARM[arm]
|
||||
cells.append(sample[(offset + run_idx) % len(sample)] + CELL_OVERHEAD_SECONDS)
|
||||
cells.append(sample[(offset + run_idx) % len(sample)])
|
||||
return cells
|
||||
|
||||
|
||||
|
|
@ -179,7 +183,9 @@ def expected_task_seconds(
|
|||
if runs < 1 or not arms:
|
||||
return 0.0
|
||||
makespan = fed_makespan if fed_pool else wave_makespan
|
||||
alignments = max(len(DURATIONS_BY_ARM[arm]) for arm in arms)
|
||||
# lcm, not max: with samples of 13 and 14, max would wrap the shorter one
|
||||
# and count its first entry twice.
|
||||
alignments = math.lcm(*(len(DURATIONS_BY_ARM[arm]) for arm in arms))
|
||||
return (
|
||||
sum(makespan(task_cells(runs, arms, offset), workers) for offset in range(alignments))
|
||||
/ alignments
|
||||
|
|
@ -189,11 +195,18 @@ def expected_task_seconds(
|
|||
def generation_seconds(
|
||||
*, task_count: int, runs: int, arms: tuple[str, ...], workers: int, fed_pool: bool
|
||||
) -> int:
|
||||
"""Whole generation: one proposer session, then the tasks back to back."""
|
||||
"""Whole generation: proposer, then the tasks back to back, plus overhead.
|
||||
|
||||
Prices a HEALTHY sweep. A run whose cells return unusable evidence does not
|
||||
reach this wall at all: the outage breaker aborts after
|
||||
``DEFAULT_OUTAGE_STREAK`` consecutive systemic failures, which for the
|
||||
sample's own error sequence is cell 5 of 41.
|
||||
"""
|
||||
|
||||
return round(
|
||||
PROPOSER_SECONDS
|
||||
+ task_count * expected_task_seconds(runs, arms, workers, fed_pool=fed_pool)
|
||||
+ task_count * runs * len(arms) * CELL_OVERHEAD_SECONDS
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -61,6 +61,7 @@
|
|||
"unaccounted_s": 2540.6,
|
||||
"cells": 41,
|
||||
"unique_shas": 5,
|
||||
"_note": "Everything the sweep spent outside the agent sessions: per-SHA sanitize and `analyze --pdg --index-only`, plus each cell's clone, materialize, staging, sandbox and teardown. That run predates clone templates and graph prefetch, so this is an upper bound for the current code. The split between per-SHA and per-cell is not recoverable from the artifact, so the model charges it per cell and serially - the pessimistic reading of an already-small term."
|
||||
}
|
||||
"_note": "Everything the sweep spent outside the agent sessions: per-SHA sanitize and `analyze --pdg --index-only`, plus each cell's clone, materialize, staging, sandbox and teardown. That run predates clone templates and graph prefetch, so this is an upper bound for the current code. The split between per-SHA and per-cell is not recoverable from the artifact, so the model charges it per cell and serially, outside the pool - the pessimistic reading of an already-small term."
|
||||
},
|
||||
"_breaker": "Replaying this sample's error_kind sequence through today's systemic_outage_streak trips the outage breaker at cell 5 of 41 (DEFAULT_OUTAGE_STREAK=5). The source run executed all 41, so its runner did not break on this sequence. The durations stay valid as per-cell timings; what they cannot describe is a 54-cell sweep with this failure profile, because the current code would never run one."
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue