diff --git a/eval/tests/test_workflow_bench.py b/eval/tests/test_workflow_bench.py index b6021a485..2663601f9 100644 --- a/eval/tests/test_workflow_bench.py +++ b/eval/tests/test_workflow_bench.py @@ -1,6 +1,6 @@ """Unit tests for the pure aggregation/report helpers of workflow_bench.""" -from workflow_bench.runner import aggregate, render_report, savings +from workflow_bench.runner import aggregate, parse_shortstat, render_report, savings def record(**overrides): @@ -12,6 +12,10 @@ def record(**overrides): "cost_usd": 0.5, "duration_s": 60.0, "num_turns": 10, + "diff_files": 2, + "diff_insertions": 30, + "diff_deletions": 5, + "class": "demo", "resolved": True, } base.update(overrides) @@ -33,6 +37,10 @@ def test_aggregate_takes_medians_and_counts_resolved(): "cost_usd": 0.5, "duration_s": 60.0, "num_turns": 10, + "diff_files": 2, + "diff_insertions": 30, + "diff_deletions": 5, + "class": "demo", "resolved": 2, "runs": 3, } @@ -53,15 +61,30 @@ def test_savings_handles_zero_baseline_without_dividing(): assert savings(baseline, workflow)["cost_usd"] == 0.0 -def test_render_report_emits_arm_rows_and_savings_row(): +def test_parse_shortstat_full_and_empty(): + full = parse_shortstat(" 3 files changed, 120 insertions(+), 7 deletions(-)") + assert full == {"diff_files": 3, "diff_insertions": 120, "diff_deletions": 7} + assert parse_shortstat("") == { + "diff_files": 0, + "diff_insertions": 0, + "diff_deletions": 0, + } + singular = parse_shortstat(" 1 file changed, 1 insertion(+)") + assert singular == {"diff_files": 1, "diff_insertions": 1, "diff_deletions": 0} + + +def test_render_report_emits_arm_rows_and_per_arm_savings_rows(): results = { "demo-task": { - "baseline": aggregate([record(input_tokens=2000)]), "workflow": aggregate([record(input_tokens=1000)]), + "workflow_direct": aggregate([record(input_tokens=1500)]), + "baseline": aggregate([record(input_tokens=2000)]), } } report = render_report(results) - assert "| demo-task | baseline | 1/1 | 2000 |" in report - assert "| demo-task | workflow | 1/1 | 1000 |" in report - assert "| demo-task | **savings %** | — | 50.0 |" in report + assert "| demo-task | demo | workflow | 1/1 | 1000 |" in report + assert "| demo-task | demo | baseline | 1/1 | 2000 |" in report + assert "| demo-task | demo | **workflow savings %** | — | 50.0 |" in report + assert "| demo-task | demo | **workflow_direct savings %** | — | 25.0 |" in report + assert "2/+30/−5" in report assert "results.jsonl" in report diff --git a/eval/workflow_bench/README.md b/eval/workflow_bench/README.md index 319f874b9..1d034e3b1 100644 --- a/eval/workflow_bench/README.md +++ b/eval/workflow_bench/README.md @@ -10,19 +10,24 @@ the CLI's own `--output-format json` usage report. | Arm | Sessions | Notes | | --- | --- | --- | | `workflow` | `gitnexus-plan` on the task, then `gitnexus-work` on the produced plan | The skills must be installed (`gitnexus setup`, or repo-local `.claude/skills/`) | +| `workflow_direct` | one `gitnexus-work` direct-mode session | The middle option — execution discipline without a planning pass | | `baseline` | one session with the identical task text | `--disallowedTools Skill` so it cannot borrow the workflow; same repo, same MCP tools | +| `baseline_nomcp` | like baseline, graph tools also disallowed | Separates the workflow-discipline question from the GitNexus-tools question (off by default) | -Both arms run in fresh detached git worktrees of the task's `ref`, once per +Every arm runs in a fresh detached git worktree of the task's `ref`, once per `--runs`. A per-task `verify` command decides `resolved` — token savings on a -failed task are flagged, not celebrated. The benchmark isolates the -*workflow discipline* (graph-first navigation, context ledger, plan→pack -handoff); both arms may use the GitNexus MCP tools. +failed task are flagged, not celebrated — and diff churn +(files/+insertions/−deletions vs the starting commit) is recorded as a cheap +over-engineering proxy. Task `class` labels (trivial → investigation → +cross-module) make the report readable as a routing table: the boundary where +`workflow` starts beating `workflow_direct` and `baseline` is the boundary +lfg's gate and work's direct-mode triage should encode. ## Quick start ```bash cd eval -uv run python -m workflow_bench.runner --tasks workflow_bench/tasks.example.yaml --runs 3 +uv run python -m workflow_bench.runner --tasks workflow_bench/tasks.scenarios.yaml --runs 3 ``` Output: `results/wfbench-/results.jsonl` (every run, with session @@ -42,7 +47,7 @@ uv run --with 'litellm[proxy]' litellm --config workflow_bench/free-model.litell # 2. Point the benchmark at it uv run python -m workflow_bench.runner \ - --tasks workflow_bench/tasks.example.yaml --runs 3 \ + --tasks workflow_bench/tasks.scenarios.yaml --runs 3 \ --base-url http://localhost:4000 --auth-token sk-wfbench --model free-coder ``` @@ -81,7 +86,7 @@ task, distrust the benchmark. ## Writing good tasks -See `tasks.example.yaml`. Small enough to finish headless, real enough to +See `tasks.scenarios.yaml`. Small enough to finish headless, real enough to require investigation — the workflow's savings come from *not re-reading and not re-investigating*, which trivial tasks never exercise. Prefer `verify` commands that use the repo's own npm scripts (they carry build pre-hooks). diff --git a/eval/workflow_bench/free-model.litellm.yaml b/eval/workflow_bench/free-model.litellm.yaml index 315ca635c..687f7e8c0 100644 --- a/eval/workflow_bench/free-model.litellm.yaml +++ b/eval/workflow_bench/free-model.litellm.yaml @@ -7,7 +7,7 @@ # uv run --with 'litellm[proxy]' litellm --config workflow_bench/free-model.litellm.yaml --port 4000 # # uv run python -m workflow_bench.runner \ -# --tasks workflow_bench/tasks.example.yaml \ +# --tasks workflow_bench/tasks.scenarios.yaml \ # --base-url http://localhost:4000 --auth-token sk-wfbench \ # --model free-coder # diff --git a/eval/workflow_bench/runner.py b/eval/workflow_bench/runner.py index afd9161df..c2460d231 100644 --- a/eval/workflow_bench/runner.py +++ b/eval/workflow_bench/runner.py @@ -22,6 +22,7 @@ from __future__ import annotations import argparse import json import os +import re import statistics import subprocess import tempfile @@ -48,6 +49,12 @@ WORK_PROMPT = ( "Headless run: proceed without asking; report Definition of Done status " "at the end." ) +WORK_DIRECT_PROMPT = ( + "Use the gitnexus-work skill for: {task}\n" + "Headless run: proceed without asking. The user explicitly declines a " + "separate planning pass — execute in direct mode with the skill's " + "execution discipline." +) BASELINE_PROMPT = ( "{task}\n\n" "Implement the change in this repository and verify it by running the " @@ -136,6 +143,31 @@ def remove_worktree(repo: Path, worktree: Path) -> None: ) +def parse_shortstat(text: str) -> dict[str, int]: + """Parse `git diff --shortstat` output into churn counters.""" + keys = { + "file": "diff_files", + "insertion": "diff_insertions", + "deletion": "diff_deletions", + } + out = dict.fromkeys(keys.values(), 0) + for count, word in re.findall(r"(\d+) (file|insertion|deletion)", text): + out[keys[word]] = int(count) + return out + + +def diff_churn(worktree: Path, orig_sha: str) -> dict[str, int]: + """Total churn (committed + uncommitted) vs the worktree's starting sha — + a cheap over-engineering proxy alongside pass/fail quality.""" + proc = subprocess.run( + ["git", "-C", str(worktree), "diff", "--shortstat", orig_sha], + capture_output=True, + text=True, + check=False, + ) + return parse_shortstat(proc.stdout) + + def run_verify(command: str, cwd: Path, timeout: int) -> bool: proc = subprocess.run( command, shell=True, cwd=cwd, capture_output=True, timeout=timeout, check=False @@ -183,6 +215,23 @@ def run_arm( **common, ) ) + elif arm == "workflow_direct": + sessions.append( + run_claude( + WORK_DIRECT_PROMPT.format(task=task["prompt"]), worktree, **common + ) + ) + elif arm == "baseline_nomcp": + # Isolates the workflow-discipline question from the GitNexus-tools + # question: no skills AND no graph tools. + sessions.append( + run_claude( + BASELINE_PROMPT.format(task=task["prompt"]), + worktree, + disallowed_tools=["Skill", "mcp__gitnexus"], + **common, + ) + ) else: sessions.append( run_claude( @@ -204,14 +253,18 @@ def run_arm( # ─── Pure aggregation/report helpers (unit-tested) ────────────────────────── +CHURN_FIELDS = ("diff_files", "diff_insertions", "diff_deletions") + + def aggregate(records: list[dict[str, Any]]) -> dict[str, Any]: """Median metrics + resolve rate across repeated runs of one task+arm.""" - metrics = (*USAGE_FIELDS, "cost_usd", "duration_s", "num_turns") + metrics = (*USAGE_FIELDS, "cost_usd", "duration_s", "num_turns", *CHURN_FIELDS) out: dict[str, Any] = { - m: statistics.median(r[m] for r in records) for m in metrics + m: statistics.median(r.get(m, 0) for r in records) for m in metrics } out["resolved"] = sum(1 for r in records if r["resolved"]) out["runs"] = len(records) + out["class"] = records[0].get("class", "") return out @@ -229,28 +282,31 @@ def render_report(results: dict[str, dict[str, dict[str, Any]]]) -> str: lines = [ "# gitnexus workflow benchmark", "", - "Medians across runs; savings = (baseline − workflow) / baseline.", - "A negative saving means the workflow arm spent more.", + "Medians across runs; savings rows = (baseline − arm) / baseline per arm.", + "A negative saving means that arm spent more than baseline. churn =", + "files/+insertions/−deletions vs the worktree's starting commit.", "", - "| task | arm | resolved | input | cache_create | cache_read | output | cost $ | wall s | turns |", - "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |", + "| task | class | arm | resolved | input | cache_create | cache_read | output | cost $ | wall s | turns | churn |", + "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |", ] for task_id, arms in results.items(): for arm, agg in arms.items(): lines.append( - f"| {task_id} | {arm} | {agg['resolved']}/{agg['runs']} " + f"| {task_id} | {agg['class']} | {arm} | {agg['resolved']}/{agg['runs']} " f"| {agg['input_tokens']:.0f} | {agg['cache_creation_input_tokens']:.0f} " f"| {agg['cache_read_input_tokens']:.0f} | {agg['output_tokens']:.0f} " - f"| {agg['cost_usd']:.4f} | {agg['duration_s']:.0f} | {agg['num_turns']:.0f} |" - ) - if "baseline" in arms and "workflow" in arms: - s = savings(arms["baseline"], arms["workflow"]) - lines.append( - f"| {task_id} | **savings %** | — " - f"| {s['input_tokens']} | {s['cache_creation_input_tokens']} " - f"| {s['cache_read_input_tokens']} | {s['output_tokens']} " - f"| {s['cost_usd']} | {s['duration_s']} | — |" + f"| {agg['cost_usd']:.4f} | {agg['duration_s']:.0f} | {agg['num_turns']:.0f} " + f"| {agg['diff_files']:.0f}/+{agg['diff_insertions']:.0f}/−{agg['diff_deletions']:.0f} |" ) + for arm in arms: + if arm != "baseline" and "baseline" in arms: + s = savings(arms["baseline"], arms[arm]) + lines.append( + f"| {task_id} | {arms[arm]['class']} | **{arm} savings %** | — " + f"| {s['input_tokens']} | {s['cache_creation_input_tokens']} " + f"| {s['cache_read_input_tokens']} | {s['output_tokens']} " + f"| {s['cost_usd']} | {s['duration_s']} | — | — |" + ) lines.append("") lines.append( "Session ids for every run are in results.jsonl — open the matching " @@ -266,7 +322,12 @@ def main() -> None: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--tasks", required=True, type=Path) parser.add_argument("--runs", type=int, default=1) - parser.add_argument("--arms", nargs="+", default=["workflow", "baseline"]) + parser.add_argument( + "--arms", + nargs="+", + default=["workflow", "workflow_direct", "baseline"], + choices=["workflow", "workflow_direct", "baseline", "baseline_nomcp"], + ) parser.add_argument("--claude-bin", default="claude") parser.add_argument("--timeout", type=int, default=3600, help="per session, seconds") parser.add_argument("--out", type=Path, default=None) @@ -315,10 +376,23 @@ def main() -> None: capture_output=True, timeout=600, ) + orig_sha = subprocess.run( + ["git", "-C", str(worktree), "rev-parse", "HEAD"], + capture_output=True, + text=True, + check=True, + ).stdout.strip() record = run_arm(arm, task, worktree, args) + record.update(diff_churn(worktree, orig_sha)) finally: remove_worktree(repo, worktree) - record.update({"task": task["id"], "run": run_idx}) + record.update( + { + "task": task["id"], + "class": task.get("class", ""), + "run": run_idx, + } + ) per_arm[arm].append(record) with results_path.open("a") as fh: fh.write(json.dumps(record) + "\n") diff --git a/eval/workflow_bench/tasks.example.yaml b/eval/workflow_bench/tasks.example.yaml deleted file mode 100644 index 6b54013af..000000000 --- a/eval/workflow_bench/tasks.example.yaml +++ /dev/null @@ -1,39 +0,0 @@ -# Example benchmark tasks for the gitnexus workflow benchmark. -# -# Each task needs: -# id: short slug used in reports -# repo: path to a git repo that is GitNexus-indexed (both arms may use -# the MCP tools — the benchmark isolates the WORKFLOW discipline, -# not GitNexus itself) -# ref: git ref to benchmark against (default HEAD); every run gets a -# fresh detached worktree of this ref -# prompt: the engineering task, phrased once, given verbatim to both arms -# setup: optional shell command run in the fresh worktree before the arm -# (dependency symlinks / installs so agents and verify can run) -# verify: shell command run in the worktree after the arm finishes; -# exit 0 = resolved. Prefer the repo's own npm scripts (they carry -# build pre-hooks). -# -# Good benchmark tasks are small enough to finish headless but real enough to -# require investigation — the workflow's savings come from NOT re-reading and -# NOT re-investigating, which trivial tasks never exercise. - -tasks: - - id: pdg-note-sublayer - repo: ~/GitNexus - ref: main - prompt: > - When the pdg_query MCP tool returns its "no PDG layer" note, make the - note say which sub-layer is missing (CDG vs REACHING_DEF) instead of a - generic message. Cover both modes with a unit test. - verify: cd gitnexus && npx vitest run test/unit --changed - - - id: list-repos-filter - repo: ~/GitNexus - ref: main - prompt: > - Add an optional "name_contains" filter parameter to the list_repos MCP - tool that filters repositories by case-insensitive substring match on - the repo name, with pagination totals reflecting the filtered set. - Cover with unit tests. - verify: cd gitnexus && npx vitest run test/unit --changed diff --git a/eval/workflow_bench/tasks.scenarios.yaml b/eval/workflow_bench/tasks.scenarios.yaml new file mode 100644 index 000000000..025af46ae --- /dev/null +++ b/eval/workflow_bench/tasks.scenarios.yaml @@ -0,0 +1,84 @@ +# Scenario suite for the gitnexus workflow benchmark — the ground-base matrix. +# +# Task fields: +# id: short slug used in reports +# class: task class for cost/quality routing analysis. The suite spans: +# trivial → investigation-bug → investigation-feature → cross-module +# repo: path to a git repo that is GitNexus-indexed +# ref: git ref benchmarked (fresh detached worktree per arm per run) +# setup: optional shell command run in the fresh worktree first +# (dependency symlinks / index copy so agents and verify can run) +# prompt: the engineering task, phrased once, given verbatim to every arm. +# Prescribe the test file path — that keeps `verify` deterministic. +# verify: shell command, exit 0 = resolved +# +# Arms (runner --arms): workflow (plan→work), workflow_direct (work skill, +# no plan), baseline (no skills, MCP allowed), baseline_nomcp (no skills, no +# graph tools). Comparing workflow vs workflow_direct vs baseline locates the +# task-complexity boundary where each mode pays for itself — that boundary is +# the routing rule lfg's gate and work's direct-mode triage encode. +# +# For the GitNexus repo itself, a working setup is: +# mkdir -p .gitnexus && +# cp -r /.gitnexus/gitnexus.json /.gitnexus/meta.json +# /.gitnexus/run.cjs /.gitnexus/lbug .gitnexus/; +# ln -s /node_modules node_modules && +# ln -s /gitnexus/node_modules gitnexus/node_modules && +# ln -s /gitnexus-shared/node_modules gitnexus-shared/node_modules + +tasks: + # Overhead floor — measured 2026-07-11 (see README calibration): the + # workflow is EXPECTED to lose here. Kept in the suite so regressions in + # the overhead floor stay visible. + - id: trivial-version-alias + class: trivial + repo: ~/GitNexus + ref: main + prompt: > + Add -V as a short alias for --version to the gitnexus CLI + (gitnexus/src/cli/index.ts), and cover the alias with a unit test in + gitnexus/test/unit/cli-commands.test.ts. + verify: cd gitnexus && npx tsc --noEmit && npx vitest run test/unit/cli-commands.test.ts + + # Investigation-heavy bug: requires locating the degraded-result path in + # local-backend, understanding the layer probe, and changing a contract + # message without breaking existing consumers. + - id: inv-bug-pdg-note + class: investigation-bug + repo: ~/GitNexus + ref: main + prompt: > + When the pdg_query MCP tool returns its "no PDG layer" note, make the + note say WHICH sub-layer is missing (CDG vs REACHING_DEF) instead of a + generic message, keeping the existing degraded-result contract intact. + Cover both modes with unit tests in + gitnexus/test/unit/pdg-note-sublayer.test.ts. + verify: cd gitnexus && npx tsc --noEmit && npx vitest run test/unit/pdg-note-sublayer.test.ts + + # Investigation-heavy feature: touches tool schema, backend filtering, and + # pagination totals — three seams that must stay consistent. + - id: inv-feature-list-repos-filter + class: investigation-feature + repo: ~/GitNexus + ref: main + prompt: > + Add an optional "name_contains" filter parameter to the list_repos MCP + tool: case-insensitive substring match on the repo name, with the + pagination object (total/hasMore/nextOffset) reflecting the FILTERED + set. Cover with unit tests in + gitnexus/test/unit/list-repos-name-filter.test.ts. + verify: cd gitnexus && npx tsc --noEmit && npx vitest run test/unit/list-repos-name-filter.test.ts + + # Cross-module: worker-pool + pipeline seams, concurrency-sensitive. + # The most expensive scenario — run deliberately, not by default. + - id: cross-module-parse-retry + class: cross-module + repo: ~/GitNexus + ref: main + prompt: > + Add bounded retry with backoff to the ingestion pipeline so a transient + parse-worker failure on a file is retried up to 2 times before the file + is marked failed, without retrying deterministic parse errors. Cover + the retry/no-retry decision with unit tests in + gitnexus/test/unit/parse-retry.test.ts. + verify: cd gitnexus && npx tsc --noEmit && npx vitest run test/unit/parse-retry.test.ts