..
fixtures
test(eval): run the benchmark offline against a scripted provider ( #3235 )
2026-09-09 12:34:00 +01:00
__init__.py
docs: agent development framework, GitHub templates, eval refactor ( #479 )
2026-03-25 06:48:41 +00:00
bench_fixtures.py
fix(eval): sweep evidence handling and measurement health, with guarded comparator reuse ( #3207 )
2026-09-08 09:25:45 +00:00
conftest.py
docs: agent development framework, GitHub templates, eval refactor ( #479 )
2026-03-25 06:48:41 +00:00
test_baseline_guidance.py
fix(release): gate publication on accuracy and paired evaluation ( #3503 )
2026-10-10 08:26:08 +01:00
test_ce_plugin_runtime.py
fix(release): gate publication on accuracy and paired evaluation ( #3503 )
2026-10-10 08:26:08 +01:00
test_check_test_execution.py
fix(ci): require every test to execute across the CI matrix ( #3479 )
2026-10-06 07:34:17 +03:00
test_comparator_reuse.py
fix(eval): sweep evidence handling and measurement health, with guarded comparator reuse ( #3207 )
2026-09-08 09:25:45 +00:00
test_ec2_workflows.py
fix(eval): materialize release graphs before paid sessions ( #3556 )
2026-10-10 15:24:09 +00:00
test_errors.py
docs: agent development framework, GitHub templates, eval refactor ( #479 )
2026-03-25 06:48:41 +00:00
test_evolution_runner.py
fix(release): gate publication on accuracy and paired evaluation ( #3503 )
2026-10-10 08:26:08 +01:00
test_evolution_runner_ready.py
fix(release): gate publication on accuracy and paired evaluation ( #3503 )
2026-10-10 08:26:08 +01:00
test_evolve.py
fix(release): gate publication on accuracy and paired evaluation ( #3503 )
2026-10-10 08:26:08 +01:00
test_mcp_bridge.py
feat(eval): evolve review skills against historical PRs
2026-09-04 05:32:31 +00:00
test_measure_evolution_cost.py
feat(eval): Add bounded packed-scheduler primitives and offline replay benchmarks ( #3206 )
2026-09-08 08:22:12 +01:00
test_mock_provider.py
test(eval): run the benchmark offline against a scripted provider ( #3235 )
2026-09-09 12:34:00 +01:00
test_model_gateway.py
fix(eval): require finite gateway startup budgets
2026-09-05 11:03:23 +00:00
test_offline_session_integration.py
test(eval): run the benchmark offline against a scripted provider ( #3235 )
2026-09-09 12:34:00 +01:00
test_offline_sweep_integration.py
fix(release): gate publication on accuracy and paired evaluation ( #3503 )
2026-10-10 08:26:08 +01:00
test_oracle_assets.py
fix(release): gate publication on accuracy and paired evaluation ( #3503 )
2026-10-10 08:26:08 +01:00
test_parse_run_id.py
docs: agent development framework, GitHub templates, eval refactor ( #479 )
2026-03-25 06:48:41 +00:00
test_process_control.py
fix(eval): sweep evidence handling and measurement health, with guarded comparator reuse ( #3207 )
2026-09-08 09:25:45 +00:00
test_promotion_apply.py
fix(eval): make evolution evidence valid and bounded
2026-09-05 10:08:35 +00:00
test_property_based.py
docs: agent development framework, GitHub templates, eval refactor ( #479 )
2026-03-25 06:48:41 +00:00
test_proposer_sandbox.py
fix(eval): build release dependencies offline without CUDA downloads ( #3541 )
2026-10-10 10:05:17 +01:00
test_provider_usage.py
feat(eval): record provider-native usage at the gateway instead of inferring it after translation ( #3220 )
2026-09-08 18:22:04 +01:00
test_provider_usage_capture.py
test(eval): run the benchmark offline against a scripted provider ( #3235 )
2026-09-09 12:34:00 +01:00
test_release_build.py
fix(eval): build release dependencies offline without CUDA downloads ( #3541 )
2026-10-10 10:05:17 +01:00
test_release_evaluation_smoke.py
fix(eval): materialize release graphs before paid sessions ( #3556 )
2026-10-10 15:24:09 +00:00
test_release_evidence.py
fix(release): gate publication on accuracy and paired evaluation ( #3503 )
2026-10-10 08:26:08 +01:00
test_release_gate.py
fix(release): gate publication on accuracy and paired evaluation ( #3503 )
2026-10-10 08:26:08 +01:00
test_release_preflight.py
fix(eval): materialize release graphs before paid sessions ( #3556 )
2026-10-10 15:24:09 +00:00
test_release_report.py
fix(release): gate publication on accuracy and paired evaluation ( #3503 )
2026-10-10 08:26:08 +01:00
test_reuse_round_trip.py
fix(eval): sweep evidence handling and measurement health, with guarded comparator reuse ( #3207 )
2026-09-08 09:25:45 +00:00
test_review_corpus.py
fix(eval): sweep evidence handling and measurement health, with guarded comparator reuse ( #3207 )
2026-09-08 09:25:45 +00:00
test_review_scoring.py
fix(eval): sweep evidence handling and measurement health, with guarded comparator reuse ( #3207 )
2026-09-08 09:25:45 +00:00
test_runner_hardening.py
fix(release): gate publication on accuracy and paired evaluation ( #3503 )
2026-10-10 08:26:08 +01:00
test_sanitized_graph.py
fix(eval): sweep evidence handling and measurement health, with guarded comparator reuse ( #3207 )
2026-09-08 09:25:45 +00:00
test_session_progress.py
fix(eval): sweep evidence handling and measurement health, with guarded comparator reuse ( #3207 )
2026-09-08 09:25:45 +00:00
test_sweep_finalization.py
fix(eval): sweep evidence handling and measurement health, with guarded comparator reuse ( #3207 )
2026-09-08 09:25:45 +00:00
test_task_assets.py
fix(eval): materialize release graphs before paid sessions ( #3556 )
2026-10-10 15:24:09 +00:00
test_tool_scripts.py
docs: agent development framework, GitHub templates, eval refactor ( #479 )
2026-03-25 06:48:41 +00:00
test_workflow_bench.py
fix(release): gate publication on accuracy and paired evaluation ( #3503 )
2026-10-10 08:26:08 +01:00
test_workflow_bench_evolution.py
fix(eval): make evolution evidence valid and bounded
2026-09-05 10:08:35 +00:00
test_workflow_bench_sessions.py
fix(release): gate publication on accuracy and paired evaluation ( #3503 )
2026-10-10 08:26:08 +01:00