diff --git a/cookbook/auto_router_roi_training/ABLATIONS.md b/cookbook/auto_router_roi_training/ABLATIONS.md index 19ca39b609b..4d1653b5bba 100644 --- a/cookbook/auto_router_roi_training/ABLATIONS.md +++ b/cookbook/auto_router_roi_training/ABLATIONS.md @@ -1,3 +1,5 @@ +> **Historical benchmark invalidated:** the 25-task live run exposed upstream fixes through Git objects outside the task checkout. Its quality and savings figures are withdrawn. These files retain the audit record, not evidence of routing improvements. Fresh isolated experiments are in progress + # Card and calibration ablations These comparisons hold the boundary at the original default: capability base 0.5 with step 0.1, or V2 quality gap 0.05. All coefficients were fitted on the training split. This table reports every raw card and the fixed middle regularization strength of 10 for calibrated variants; the JSON contains all strengths. No variant is chosen using these held-out results diff --git a/cookbook/auto_router_roi_training/BENCHMARK.md b/cookbook/auto_router_roi_training/BENCHMARK.md index 317e11186a1..13ec04b8136 100644 --- a/cookbook/auto_router_roi_training/BENCHMARK.md +++ b/cookbook/auto_router_roi_training/BENCHMARK.md @@ -1,3 +1,5 @@ +> **Historical benchmark invalidated:** the 25-task live run exposed upstream fixes through Git objects outside the task checkout. Its quality and savings figures are withdrawn. These files retain the audit record, not evidence of routing improvements. Fresh isolated experiments are in progress + # Auto Router training experiment Profiles were fitted on 83 DeepSWE tasks and selected on 30 repository-disjoint validation tasks before inspecting live grades. The live comparison uses 25 native-image-eligible SWE-bench Verified tasks and four current solver models at high effort. Each solver runs once per task; frozen task-pinned policies reuse those attempts and add their judge cost diff --git a/cookbook/auto_router_roi_training/FINDINGS.md b/cookbook/auto_router_roi_training/FINDINGS.md index 64a052b27ac..07acc442639 100644 --- a/cookbook/auto_router_roi_training/FINDINGS.md +++ b/cookbook/auto_router_roi_training/FINDINGS.md @@ -1,3 +1,5 @@ +> **Historical benchmark invalidated:** the 25-task live run exposed upstream fixes through Git objects outside the task checkout. Its quality and savings figures are withdrawn. These files retain the audit record, not evidence of routing improvements. Fresh isolated experiments are in progress + # What the training experiment showed Both implementations now support runnable, opt-in trained profiles. Training covered three cards, raw probabilities, per-model calibration, task-dependent calibration, and pair-specific boundaries, including combinations. All fitting and policy selection used the DeepSWE training and validation splits. The 25 fresh SWE-bench tasks were used only for evaluation diff --git a/cookbook/auto_router_roi_training/README.md b/cookbook/auto_router_roi_training/README.md index 37f78b89213..e19b175724a 100644 --- a/cookbook/auto_router_roi_training/README.md +++ b/cookbook/auto_router_roi_training/README.md @@ -1,3 +1,5 @@ +> **Historical benchmark invalidated:** the 25-task live run exposed upstream fixes through Git objects outside the task checkout. Its quality and savings figures are withdrawn. These files retain the audit record, not evidence of routing improvements. Fresh isolated experiments are in progress + # Experimental Auto Router training snapshots These opt-in profiles compare the original cards, a research-informed card rewrite, and cards with training-derived capability priors. They also compare raw probabilities, per-model logit calibration, and regularized task-dependent logit calibration. The router makes one judge call; all learned probability adjustments and threshold comparisons run locally diff --git a/cookbook/auto_router_roi_training/ablation_diagnostics.json b/cookbook/auto_router_roi_training/ablation_diagnostics.json index a18d5f5b4f0..c0bc9c93b87 100644 --- a/cookbook/auto_router_roi_training/ablation_diagnostics.json +++ b/cookbook/auto_router_roi_training/ablation_diagnostics.json @@ -2395,5 +2395,10 @@ "lost_capable_successes": 0, "gained_efficient_successes": 0 } - ] + ], + "validation_status": { + "live_benchmark_valid": false, + "reason": "Solver-visible Git history exposed upstream fixes; live quality and savings claims withdrawn", + "replacement": "Fresh isolated training and evaluation in progress" + } } diff --git a/cookbook/auto_router_roi_training/benchmark_results.json b/cookbook/auto_router_roi_training/benchmark_results.json index 12c9ec87f5c..d1396e013a8 100644 --- a/cookbook/auto_router_roi_training/benchmark_results.json +++ b/cookbook/auto_router_roi_training/benchmark_results.json @@ -4779,5 +4779,10 @@ } } } - ] + ], + "validation_status": { + "live_benchmark_valid": false, + "reason": "Solver-visible Git history exposed upstream fixes; live quality and savings claims withdrawn", + "replacement": "Fresh isolated training and evaluation in progress" + } } diff --git a/cookbook/auto_router_roi_training/manifest.json b/cookbook/auto_router_roi_training/manifest.json index 99fe504f8f7..4bbe557c0a6 100644 --- a/cookbook/auto_router_roi_training/manifest.json +++ b/cookbook/auto_router_roi_training/manifest.json @@ -51,5 +51,10 @@ "ABLATIONS.md", "ablation_diagnostics.json" ] + }, + "validation_status": { + "live_benchmark_valid": false, + "reason": "Solver-visible Git history exposed upstream fixes; live quality and savings claims withdrawn", + "replacement": "Fresh isolated training and evaluation in progress" } }