rerun-bench 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. rerun_bench-0.1.0/.gitignore +11 -0
  2. rerun_bench-0.1.0/CHANGELOG.md +59 -0
  3. rerun_bench-0.1.0/LICENSE +21 -0
  4. rerun_bench-0.1.0/PKG-INFO +262 -0
  5. rerun_bench-0.1.0/README.md +240 -0
  6. rerun_bench-0.1.0/docs/METRICS.md +99 -0
  7. rerun_bench-0.1.0/docs/demo.gif +0 -0
  8. rerun_bench-0.1.0/docs/demo.tape +32 -0
  9. rerun_bench-0.1.0/docs/pilot-2026-10-03/README.md +101 -0
  10. rerun_bench-0.1.0/docs/pilot-2026-10-03/report.html +85 -0
  11. rerun_bench-0.1.0/docs/pilot-2026-10-03/report.json +1554 -0
  12. rerun_bench-0.1.0/docs/pilot-2026-10-03/report.md +27 -0
  13. rerun_bench-0.1.0/pyproject.toml +59 -0
  14. rerun_bench-0.1.0/src/rerun_bench/__init__.py +3 -0
  15. rerun_bench-0.1.0/src/rerun_bench/__main__.py +3 -0
  16. rerun_bench-0.1.0/src/rerun_bench/adapters/__init__.py +31 -0
  17. rerun_bench-0.1.0/src/rerun_bench/adapters/base.py +190 -0
  18. rerun_bench-0.1.0/src/rerun_bench/adapters/claude.py +135 -0
  19. rerun_bench-0.1.0/src/rerun_bench/adapters/codex.py +112 -0
  20. rerun_bench-0.1.0/src/rerun_bench/adapters/mock.py +111 -0
  21. rerun_bench-0.1.0/src/rerun_bench/adapters/opencode.py +74 -0
  22. rerun_bench-0.1.0/src/rerun_bench/cli.py +235 -0
  23. rerun_bench-0.1.0/src/rerun_bench/metrics.py +216 -0
  24. rerun_bench-0.1.0/src/rerun_bench/report.py +316 -0
  25. rerun_bench-0.1.0/src/rerun_bench/runner.py +278 -0
  26. rerun_bench-0.1.0/src/rerun_bench/tasks.py +150 -0
  27. rerun_bench-0.1.0/src/rerun_bench/workspace.py +122 -0
  28. rerun_bench-0.1.0/tasks/add-cli-flag/solution/wc.py +33 -0
  29. rerun_bench-0.1.0/tasks/add-cli-flag/task.toml +9 -0
  30. rerun_bench-0.1.0/tasks/add-cli-flag/verify.py +48 -0
  31. rerun_bench-0.1.0/tasks/add-cli-flag/workspace/wc.py +27 -0
  32. rerun_bench-0.1.0/tasks/edit-config/solution/config/app.toml +14 -0
  33. rerun_bench-0.1.0/tasks/edit-config/task.toml +9 -0
  34. rerun_bench-0.1.0/tasks/edit-config/verify.py +36 -0
  35. rerun_bench-0.1.0/tasks/edit-config/workspace/config/app.prod.toml +14 -0
  36. rerun_bench-0.1.0/tasks/edit-config/workspace/config/app.toml +14 -0
  37. rerun_bench-0.1.0/tasks/fix-failing-test/solution/stats.py +23 -0
  38. rerun_bench-0.1.0/tasks/fix-failing-test/task.toml +8 -0
  39. rerun_bench-0.1.0/tasks/fix-failing-test/verify.py +54 -0
  40. rerun_bench-0.1.0/tasks/fix-failing-test/workspace/stats.py +21 -0
  41. rerun_bench-0.1.0/tasks/fix-failing-test/workspace/test_stats.py +22 -0
  42. rerun_bench-0.1.0/tasks/follow-agents-md/solution/CHANGES.md +10 -0
  43. rerun_bench-0.1.0/tasks/follow-agents-md/solution/textutils.py +19 -0
  44. rerun_bench-0.1.0/tasks/follow-agents-md/task.toml +9 -0
  45. rerun_bench-0.1.0/tasks/follow-agents-md/verify.py +55 -0
  46. rerun_bench-0.1.0/tasks/follow-agents-md/workspace/AGENTS.md +8 -0
  47. rerun_bench-0.1.0/tasks/follow-agents-md/workspace/CHANGES.md +8 -0
  48. rerun_bench-0.1.0/tasks/follow-agents-md/workspace/textutils.py +13 -0
  49. rerun_bench-0.1.0/tasks/implement-lru-cache/solution/lru.py +44 -0
  50. rerun_bench-0.1.0/tasks/implement-lru-cache/task.toml +8 -0
  51. rerun_bench-0.1.0/tasks/implement-lru-cache/verify.py +58 -0
  52. rerun_bench-0.1.0/tasks/implement-lru-cache/workspace/lru.py +31 -0
  53. rerun_bench-0.1.0/tasks/implement-slugify/solution/textkit/slug.py +30 -0
  54. rerun_bench-0.1.0/tasks/implement-slugify/task.toml +8 -0
  55. rerun_bench-0.1.0/tasks/implement-slugify/verify.py +49 -0
  56. rerun_bench-0.1.0/tasks/implement-slugify/workspace/textkit/__init__.py +0 -0
  57. rerun_bench-0.1.0/tasks/implement-slugify/workspace/textkit/slug.py +21 -0
  58. rerun_bench-0.1.0/tasks/minimal-fix/solution/legacy.py +33 -0
  59. rerun_bench-0.1.0/tasks/minimal-fix/task.toml +10 -0
  60. rerun_bench-0.1.0/tasks/minimal-fix/verify.py +77 -0
  61. rerun_bench-0.1.0/tasks/minimal-fix/workspace/legacy.py +33 -0
  62. rerun_bench-0.1.0/tasks/minimal-fix/workspace/report.py +10 -0
  63. rerun_bench-0.1.0/tasks/multi-file-rename/solution/inventory/__init__.py +5 -0
  64. rerun_bench-0.1.0/tasks/multi-file-rename/solution/inventory/cli.py +18 -0
  65. rerun_bench-0.1.0/tasks/multi-file-rename/solution/inventory/orders.py +14 -0
  66. rerun_bench-0.1.0/tasks/multi-file-rename/solution/inventory/pricing.py +12 -0
  67. rerun_bench-0.1.0/tasks/multi-file-rename/task.toml +9 -0
  68. rerun_bench-0.1.0/tasks/multi-file-rename/verify.py +51 -0
  69. rerun_bench-0.1.0/tasks/multi-file-rename/workspace/inventory/__init__.py +5 -0
  70. rerun_bench-0.1.0/tasks/multi-file-rename/workspace/inventory/cli.py +18 -0
  71. rerun_bench-0.1.0/tasks/multi-file-rename/workspace/inventory/orders.py +14 -0
  72. rerun_bench-0.1.0/tasks/multi-file-rename/workspace/inventory/pricing.py +12 -0
  73. rerun_bench-0.1.0/tasks/refactor-extract-helper/solution/pricing.py +32 -0
  74. rerun_bench-0.1.0/tasks/refactor-extract-helper/task.toml +9 -0
  75. rerun_bench-0.1.0/tasks/refactor-extract-helper/verify.py +63 -0
  76. rerun_bench-0.1.0/tasks/refactor-extract-helper/workspace/pricing.py +32 -0
  77. rerun_bench-0.1.0/tasks/write-tests/mutants/accepts_empty.py +20 -0
  78. rerun_bench-0.1.0/tasks/write-tests/mutants/drops_seconds.py +20 -0
  79. rerun_bench-0.1.0/tasks/write-tests/mutants/hours_as_minutes.py +20 -0
  80. rerun_bench-0.1.0/tasks/write-tests/mutants/minutes_wrong_factor.py +20 -0
  81. rerun_bench-0.1.0/tasks/write-tests/mutants/no_strip.py +19 -0
  82. rerun_bench-0.1.0/tasks/write-tests/solution/tests/test_duration.py +27 -0
  83. rerun_bench-0.1.0/tasks/write-tests/task.toml +11 -0
  84. rerun_bench-0.1.0/tasks/write-tests/verify.py +52 -0
  85. rerun_bench-0.1.0/tasks/write-tests/workspace/duration.py +20 -0
  86. rerun_bench-0.1.0/tasks/write-tests/workspace/tests/__init__.py +0 -0
  87. rerun_bench-0.1.0/tests/__init__.py +0 -0
  88. rerun_bench-0.1.0/tests/conftest.py +13 -0
  89. rerun_bench-0.1.0/tests/test_adapters.py +377 -0
  90. rerun_bench-0.1.0/tests/test_cli.py +238 -0
  91. rerun_bench-0.1.0/tests/test_metrics.py +138 -0
  92. rerun_bench-0.1.0/tests/test_tasks.py +100 -0
  93. rerun_bench-0.1.0/tests/test_workspace.py +51 -0
@@ -0,0 +1,11 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .venv/
4
+ dist/
5
+ build/
6
+ *.egg-info/
7
+ .pytest_cache/
8
+ .ruff_cache/
9
+ results/
10
+ .coverage
11
+ .DS_Store
@@ -0,0 +1,59 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented here. The format follows
4
+ [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and the project uses
5
+ [Semantic Versioning](https://semver.org/).
6
+
7
+ ## Unreleased
8
+
9
+ ### Changed
10
+
11
+ - Renamed the project from `rerunbench` to `rerun-bench`. The repository is now
12
+ github.com/Abelo9996/rerun-bench, the command is `rerun-bench`, the Python import package
13
+ is `rerun_bench`, and the agent skill lives in `skills/rerun-bench/`. Environment
14
+ variables are now `RERUN_BENCH_TASKS_DIR` and `RERUN_BENCH_TASK_DIR`, and result
15
+ metadata records `rerun_bench_version`.
16
+ - `codex` runs pass `--ignore-user-config` by default (`--agent-opt isolate=0` to disable),
17
+ so a personal `config.toml` (model, effort, plugins, notify hooks) does not leak into the
18
+ measurement.
19
+ - Recorded verifier output has the workspace, task and home directory paths replaced with
20
+ `<workspace>`, `<task>` and `~`, so result directories can be published as they are.
21
+
22
+ ### Added
23
+
24
+ - `run --resume` continues an interrupted `--run-id`, running only the missing task and run
25
+ pairs. A truncated last line in `runs.jsonl` is dropped instead of breaking the report.
26
+ - `--agent-opt bin=<path>` runs a specific build of an agent CLI.
27
+ - `--agent-opt effort=<level>` for `claude` (`--effort`) and `codex`
28
+ (`model_reasoning_effort`).
29
+ - Run records include `agent_stdout_tail` and `agent_stderr_tail` (paths scrubbed), so a
30
+ run that fails without changing anything can be diagnosed from the result files.
31
+ - Pilot results for Claude Code and Codex CLI in `docs/pilot-2026-10-03/`.
32
+ - JSON report: `median_cost_usd`, `tokens_median`, `output_tokens_median` and
33
+ `wall_time_total_s`.
34
+
35
+ ### Fixed
36
+
37
+ - `claude`: tokens are summed over every model in `modelUsage`, so they cover the same
38
+ calls as `total_cost_usd`; previously only the main conversation's `usage` was counted and
39
+ subagent or background-model tokens were missed.
40
+ - `claude`: when started from inside a Claude Code session, the parent session's
41
+ environment variables (`CLAUDECODE`, `CLAUDE_CODE_SESSION_ID`, `CLAUDE_EFFORT` and
42
+ others) are no longer inherited by the measured session.
43
+ - `codex`: transient `error` events that Codex retries ("Reconnecting... 2/5") no longer
44
+ mark a successful run as an error; only `turn.failed`, or errors with no completed turn,
45
+ do.
46
+ - `codex`: `cache_write_input_tokens` (Codex CLI 0.160) is read and split out of input
47
+ tokens instead of being counted as uncached input.
48
+ - `codex`: a cached-token price of `0` is no longer replaced by the input price.
49
+
50
+ ## 0.1.0 - 2026-10-03
51
+
52
+ - First release.
53
+ - `rerun-bench list`, `run`, `report` (md, html, json) and `verify-tasks` commands.
54
+ - Adapters: `claude`, `codex`, `opencode`, and a deterministic, seeded `mock`.
55
+ - Ten offline, deterministic tasks with hidden verifiers and reference solutions.
56
+ - Metrics: pass rate with Wilson and task-bootstrap intervals, pass@k, pass^k, flip rate,
57
+ flaky task fraction, cost, token and wall-time spread (within-task CV), cost per success,
58
+ and approach similarity of passing diffs.
59
+ - Self-contained static HTML report.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Abel Yagubyan
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,262 @@
1
+ Metadata-Version: 2.5
2
+ Name: rerun-bench
3
+ Version: 0.1.0
4
+ Summary: Run the same coding task N times per agent and measure how consistent and how expensive it is.
5
+ Project-URL: Homepage, https://github.com/Abelo9996/rerun-bench
6
+ Project-URL: Issues, https://github.com/Abelo9996/rerun-bench/issues
7
+ Author: Abel Yagubyan
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: benchmark,coding-agents,consistency,llm-evaluation,reliability
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Environment :: Console
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Topic :: Software Development :: Testing
20
+ Requires-Python: >=3.11
21
+ Description-Content-Type: text/markdown
22
+
23
+ # rerun-bench
24
+
25
+ [![CI](https://github.com/Abelo9996/rerun-bench/actions/workflows/ci.yml/badge.svg)](https://github.com/Abelo9996/rerun-bench/actions/workflows/ci.yml)
26
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
27
+ [![Python 3.11+](https://img.shields.io/badge/python-3.11%2B-blue.svg)](pyproject.toml)
28
+
29
+ ![rerun-bench running the free mock agent 5 times on each of 10 tasks, then printing each task's pass and fail sequence, pass rate, flip rate and cost spread](docs/demo.gif)
30
+
31
+ Run the same coding task N times per agent and find out how often it succeeds, how often it
32
+ flips between pass and fail, and how much the bill varies from run to run.
33
+
34
+ Most coding-agent benchmarks (SWE-bench, Terminal-Bench and its harbor harness) report a
35
+ success rate from one attempt per task. That number hides what you live with day to day: the
36
+ agent that fixed the bug on Monday fails the same task on Tuesday, at twice the token cost.
37
+ rerun-bench runs a fixed suite of small, verifiable tasks several times per agent, model and CLI
38
+ version, then reports reliability (pass^k, flip rate) and cost spread (coefficient of
39
+ variation) next to the pass rate.
40
+
41
+ ## Quickstart (free, about 30 seconds)
42
+
43
+ The `mock` adapter simulates an agent with a configurable pass probability and token usage. It
44
+ spends nothing and needs no API key, so you can see the whole pipeline before pointing it at a
45
+ paid agent.
46
+
47
+ ```sh
48
+ uvx --from git+https://github.com/Abelo9996/rerun-bench rerun-bench list
49
+ uvx --from git+https://github.com/Abelo9996/rerun-bench rerun-bench run --agent mock --tasks all --runs 5 --out results/
50
+ uvx --from git+https://github.com/Abelo9996/rerun-bench rerun-bench report results/ --format html -o report.html
51
+ ```
52
+
53
+ Or install once with `uv tool install git+https://github.com/Abelo9996/rerun-bench` and drop
54
+ the `uvx --from ...` prefix.
55
+
56
+ ## Run real agents
57
+
58
+ Supported CLIs, each driven headlessly in a fresh temporary copy of the task workspace:
59
+
60
+ | Agent | Command rerun-bench runs | Cost reported by the CLI |
61
+ |---|---|---|
62
+ | `claude` (Claude Code) | `claude -p <prompt> --output-format json --permission-mode bypassPermissions` | yes (`total_cost_usd`) |
63
+ | `codex` (OpenAI Codex CLI) | `codex exec --json --ephemeral --ignore-user-config --sandbox workspace-write --cd <ws> <prompt>` | tokens only; pass prices with `--agent-opt` to get dollars |
64
+ | `opencode` | `opencode run --format json <prompt>` | yes (per step) |
65
+
66
+ ```sh
67
+ rerun-bench run --agent claude --model sonnet --tasks all --runs 5 --out results/ --yes
68
+ rerun-bench run --agent codex --model <model> --runs 5 --out results/ --yes \
69
+ --agent-opt usd_per_mtok_in=1.25 --agent-opt usd_per_mtok_out=10 --agent-opt usd_per_mtok_cached=0.125
70
+ rerun-bench report results/ --format md
71
+ ```
72
+
73
+ The `usd_per_mtok_*` values are placeholders; use the published prices of the model you run.
74
+
75
+ Each run records wall time, exit status, token usage and cost (when the CLI reports them),
76
+ the CLI version, the model, and the final diff. A value the CLI does not report is stored as
77
+ `null`, never as zero.
78
+
79
+ **Cost warning.** Real runs spend your API credit or subscription quota: a full suite at
80
+ `--runs 5` is 50 agent sessions. rerun-bench refuses to start a real agent without `--yes`, and
81
+ prints the run count first. Start with `--tasks edit-config --runs 2`. The agent runs with
82
+ file-edit and shell permissions inside a temp directory; treat it like any other unattended
83
+ agent session.
84
+
85
+ Personal configuration is kept out of the measurement by default. For `claude`, rerun-bench
86
+ loads only project and local settings and ignores MCP servers outside `--mcp-config`, so your
87
+ hooks, plugins and MCP servers do not apply. For `codex`, it passes `--ignore-user-config`, so
88
+ the model, reasoning effort, plugins and notify hooks in your `config.toml` do not apply (auth
89
+ still works). `--agent-opt isolate=0` turns this off for either agent. When rerun-bench itself
90
+ runs inside a Claude Code session, that session's environment variables (`CLAUDECODE`,
91
+ `CLAUDE_CODE_SESSION_ID` and similar) are removed before starting the measured `claude`.
92
+
93
+ `codex exec --json` does not report which model it ran, so pass `--model` if you want the
94
+ model recorded; otherwise the report shows `default`.
95
+
96
+ Other useful flags: `--jobs 4` (parallel runs), `--tasks tag:refactor` or `--tasks a,b`,
97
+ `--keep-workspaces` (inspect what the agent left behind), `--seed` (mock only),
98
+ `--agent-opt bin=/path/to/cli` (run a specific build of the CLI), `--agent-opt effort=high`
99
+ (`claude --effort` or Codex `model_reasoning_effort`).
100
+
101
+ Long runs can be interrupted and continued: name the run with `--run-id` and add `--resume`
102
+ to run only the task and run pairs that `runs.jsonl` does not have yet.
103
+
104
+ ```sh
105
+ rerun-bench run --agent claude --tasks all --runs 3 --out results/ --run-id claude-pilot --yes
106
+ # interrupted; later:
107
+ rerun-bench run --agent claude --tasks all --runs 3 --out results/ --run-id claude-pilot --yes --resume
108
+ ```
109
+
110
+ ## Pilot results
111
+
112
+ A first run against real CLIs on 2026-10-03: all 10 tasks, 3 runs each, Claude Code 2.1.288
113
+ (default model, reported as `claude-opus-5-5`) and Codex CLI 0.160.0 (`gpt-6-luna`), on
114
+ macOS arm64. Full setup, per-task outcomes, raw run records and diffs:
115
+ [docs/pilot-2026-10-03](docs/pilot-2026-10-03/README.md).
116
+
117
+ | Agent / model | Pass rate [Wilson 95% CI] | pass^3 | Flip rate | Median cost/run | Median tokens/run | Median wall time |
118
+ |---|---|---|---|---|---|---|
119
+ | claude / claude-opus-5-5 | 30/30, 100% [89, 100] | 100% | 0% | $0.0886 | 52,017 | 12.6 s |
120
+ | codex / gpt-6-luna | 28/30, 93% [79, 98] | 80% | 13% | not reported | 56,629 | 16.1 s |
121
+
122
+ n = 3 per task is a pilot, not a leaderboard. The pass-rate intervals overlap, so these
123
+ runs do not establish a difference between the two agents. Claude Code's cost is its own
124
+ list-price estimate; Codex reports tokens only.
125
+
126
+ ## Example report
127
+
128
+ Two mock profiles, 10 tasks, 5 runs each (`rerun-bench report results/`):
129
+
130
+ | Agent / model | Runs/task | Pass rate [95% CI] | pass@k | pass^k | Flip rate | Flaky tasks | Cost/run | Cost CV |
131
+ |---|---|---|---|---|---|---|---|---|
132
+ | mock / mock-steady | 5 | 86% [74, 93] | 100% | 40% | 26% | 60% | $0.0601 | 0.19 |
133
+ | mock / mock-flaky | 5 | 60% [46, 72] | 100% | 0% | 50% | 100% | $0.0906 | 0.46 |
134
+
135
+ | Task | mock / mock-steady | mock / mock-flaky |
136
+ |---|---|---|
137
+ | fix-failing-test | `PPPPP` 100%, flip 0%, cost CV 0.13 | `FFFPF` 20%, flip 40%, cost CV 0.49 |
138
+ | minimal-fix | `PPPPF` 80%, flip 40%, cost CV 0.16 | `PPFPF` 60%, flip 60%, cost CV 0.46 |
139
+
140
+ Both profiles reach pass@5 = 100%: given five tries, each solves every task at least once.
141
+ Only pass^5 and the flip rate separate them. The HTML report (`--format html`) is one static
142
+ file with inline CSS and JS, a sortable leaderboard, and a per-task grid of run outcomes.
143
+
144
+ ## Metrics
145
+
146
+ Full definitions, estimators and caveats: [docs/METRICS.md](docs/METRICS.md).
147
+
148
+ | Metric | What it answers |
149
+ |---|---|
150
+ | Pass rate + Wilson 95% CI | How often does a run pass the hidden verifier? |
151
+ | Task-bootstrap 95% CI | Same, with uncertainty over which tasks were sampled (JSON report). |
152
+ | pass@k | Chance that at least one of k runs passes (unbiased estimator). |
153
+ | pass^k | Chance that all k runs pass. The number to watch if you run once and trust the result. |
154
+ | Flip rate | Chance that two runs of the same task disagree, 2c(n-c)/(n(n-1)). |
155
+ | Flaky tasks | Share of tasks with both passes and fails. |
156
+ | Cost / tokens / wall-time CV | Run-to-run spread within a task (std / mean), averaged over tasks. |
157
+ | Cost per success | Total cost divided by passing runs. |
158
+ | Approach similarity | Mean pairwise Jaccard of changed lines among passing runs. 1.0 means the same edit every time. |
159
+
160
+ Only the task's verifier decides pass or fail. The agent's exit code and its own claims of
161
+ success are recorded but not scored.
162
+
163
+ ## Task suite
164
+
165
+ `rerun-bench list` shows the bundled tasks:
166
+
167
+ | Task | What it tests |
168
+ |---|---|
169
+ | `fix-failing-test` | Fix the bug behind a failing unit test without editing the test |
170
+ | `implement-slugify` | Implement a function exactly to a docstring spec |
171
+ | `implement-lru-cache` | Implement a small data structure to spec |
172
+ | `refactor-extract-helper` | Extract duplicated logic; behavior checked on a grid of inputs |
173
+ | `follow-agents-md` | Add a function; the prompt does not mention the repo's AGENTS.md rules, the verifier checks them |
174
+ | `edit-config` | Three precise TOML edits; a production config next to it must stay untouched |
175
+ | `multi-file-rename` | Rename a function across a package, no alias left behind |
176
+ | `minimal-fix` | One-line bug in deliberately dated code; any cleanup outside the function fails |
177
+ | `add-cli-flag` | Add a flag without changing default output |
178
+ | `write-tests` | Write tests that pass on the real code and catch five injected bugs (mutation testing) |
179
+
180
+ Every task is offline, deterministic, and uses only the Python standard library, so it runs
181
+ the same on Linux, macOS and Windows.
182
+
183
+ ## Add a task
184
+
185
+ ```
186
+ tasks/<id>/
187
+ task.toml id, title, prompt, timeout (seconds), tags
188
+ workspace/ the files the agent starts with
189
+ verify.py exit 0 = pass; runs with cwd = the agent's workspace; never shown to the agent
190
+ solution/ reference solution, copied over workspace/ by the test suite
191
+ ```
192
+
193
+ ```toml
194
+ id = "my-task"
195
+ title = "One line describing the task"
196
+ prompt = """
197
+ What you would type to the agent.
198
+ """
199
+ timeout = 600
200
+ tags = ["bugfix", "python"]
201
+ ```
202
+
203
+ Then check it: `rerun-bench --tasks-dir tasks verify-tasks --tasks my-task -v` must print `ok`
204
+ (the untouched workspace fails, the reference solution passes), and `uv run pytest` picks the
205
+ new task up automatically. Rules for verifiers are in [CONTRIBUTING.md](CONTRIBUTING.md).
206
+
207
+ ## Add an adapter
208
+
209
+ Subclass `Adapter` in `src/rerun_bench/adapters/`, implement two pure methods, and register it
210
+ in `ADAPTERS` in `src/rerun_bench/adapters/__init__.py`:
211
+
212
+ ```python
213
+ class MyAgentAdapter(Adapter):
214
+ name = "myagent"
215
+ binary = "myagent"
216
+
217
+ def build_command(self, prompt: str, workspace: Path) -> list[str]:
218
+ return [self.binary, "run", "--json", prompt]
219
+
220
+ def parse_output(self, stdout: str, stderr: str) -> Usage:
221
+ # Fill tokens, cost and model; leave a field None when the CLI does not report it.
222
+ data = json.loads(stdout)
223
+ return Usage(output_tokens=data.get("output_tokens"), cost_usd=data.get("cost"))
224
+ ```
225
+
226
+ The base class handles the subprocess, timeout, wall time and `--version`. Test both methods
227
+ against a captured sample of the CLI's output (see `tests/test_adapters.py`); the test suite
228
+ never calls a real agent.
229
+
230
+ ## Agent skill
231
+
232
+ `skills/rerun-bench/SKILL.md` teaches a coding agent to run the benchmark and add tasks:
233
+
234
+ ```sh
235
+ npx skills add Abelo9996/rerun-bench
236
+ ```
237
+
238
+ ## Roadmap
239
+
240
+ - Public leaderboard, refreshed on model launch days, built from the HTML report.
241
+ - Version-over-version tracking: the same model under successive CLI releases, with
242
+ per-task significance tests.
243
+ - Paired comparisons between two result sets (Fisher exact per task, task-level bootstrap
244
+ for the suite).
245
+ - More tasks in other languages, kept small, offline and deterministic.
246
+
247
+ ## Related projects
248
+
249
+ - [nerf-watch](https://github.com/Abelo9996/nerf-watch): detects silent model and cost changes
250
+ from your local agent logs.
251
+ - [snap-back](https://github.com/Abelo9996/snap-back): undo for any coding agent.
252
+
253
+ ## Development
254
+
255
+ ```sh
256
+ uv sync
257
+ uv run pytest
258
+ uv run ruff check . && uv run ruff format --check .
259
+ uv run rerun-bench verify-tasks
260
+ ```
261
+
262
+ MIT licensed. See [LICENSE](LICENSE).
@@ -0,0 +1,240 @@
1
+ # rerun-bench
2
+
3
+ [![CI](https://github.com/Abelo9996/rerun-bench/actions/workflows/ci.yml/badge.svg)](https://github.com/Abelo9996/rerun-bench/actions/workflows/ci.yml)
4
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
5
+ [![Python 3.11+](https://img.shields.io/badge/python-3.11%2B-blue.svg)](pyproject.toml)
6
+
7
+ ![rerun-bench running the free mock agent 5 times on each of 10 tasks, then printing each task's pass and fail sequence, pass rate, flip rate and cost spread](docs/demo.gif)
8
+
9
+ Run the same coding task N times per agent and find out how often it succeeds, how often it
10
+ flips between pass and fail, and how much the bill varies from run to run.
11
+
12
+ Most coding-agent benchmarks (SWE-bench, Terminal-Bench and its harbor harness) report a
13
+ success rate from one attempt per task. That number hides what you live with day to day: the
14
+ agent that fixed the bug on Monday fails the same task on Tuesday, at twice the token cost.
15
+ rerun-bench runs a fixed suite of small, verifiable tasks several times per agent, model and CLI
16
+ version, then reports reliability (pass^k, flip rate) and cost spread (coefficient of
17
+ variation) next to the pass rate.
18
+
19
+ ## Quickstart (free, about 30 seconds)
20
+
21
+ The `mock` adapter simulates an agent with a configurable pass probability and token usage. It
22
+ spends nothing and needs no API key, so you can see the whole pipeline before pointing it at a
23
+ paid agent.
24
+
25
+ ```sh
26
+ uvx --from git+https://github.com/Abelo9996/rerun-bench rerun-bench list
27
+ uvx --from git+https://github.com/Abelo9996/rerun-bench rerun-bench run --agent mock --tasks all --runs 5 --out results/
28
+ uvx --from git+https://github.com/Abelo9996/rerun-bench rerun-bench report results/ --format html -o report.html
29
+ ```
30
+
31
+ Or install once with `uv tool install git+https://github.com/Abelo9996/rerun-bench` and drop
32
+ the `uvx --from ...` prefix.
33
+
34
+ ## Run real agents
35
+
36
+ Supported CLIs, each driven headlessly in a fresh temporary copy of the task workspace:
37
+
38
+ | Agent | Command rerun-bench runs | Cost reported by the CLI |
39
+ |---|---|---|
40
+ | `claude` (Claude Code) | `claude -p <prompt> --output-format json --permission-mode bypassPermissions` | yes (`total_cost_usd`) |
41
+ | `codex` (OpenAI Codex CLI) | `codex exec --json --ephemeral --ignore-user-config --sandbox workspace-write --cd <ws> <prompt>` | tokens only; pass prices with `--agent-opt` to get dollars |
42
+ | `opencode` | `opencode run --format json <prompt>` | yes (per step) |
43
+
44
+ ```sh
45
+ rerun-bench run --agent claude --model sonnet --tasks all --runs 5 --out results/ --yes
46
+ rerun-bench run --agent codex --model <model> --runs 5 --out results/ --yes \
47
+ --agent-opt usd_per_mtok_in=1.25 --agent-opt usd_per_mtok_out=10 --agent-opt usd_per_mtok_cached=0.125
48
+ rerun-bench report results/ --format md
49
+ ```
50
+
51
+ The `usd_per_mtok_*` values are placeholders; use the published prices of the model you run.
52
+
53
+ Each run records wall time, exit status, token usage and cost (when the CLI reports them),
54
+ the CLI version, the model, and the final diff. A value the CLI does not report is stored as
55
+ `null`, never as zero.
56
+
57
+ **Cost warning.** Real runs spend your API credit or subscription quota: a full suite at
58
+ `--runs 5` is 50 agent sessions. rerun-bench refuses to start a real agent without `--yes`, and
59
+ prints the run count first. Start with `--tasks edit-config --runs 2`. The agent runs with
60
+ file-edit and shell permissions inside a temp directory; treat it like any other unattended
61
+ agent session.
62
+
63
+ Personal configuration is kept out of the measurement by default. For `claude`, rerun-bench
64
+ loads only project and local settings and ignores MCP servers outside `--mcp-config`, so your
65
+ hooks, plugins and MCP servers do not apply. For `codex`, it passes `--ignore-user-config`, so
66
+ the model, reasoning effort, plugins and notify hooks in your `config.toml` do not apply (auth
67
+ still works). `--agent-opt isolate=0` turns this off for either agent. When rerun-bench itself
68
+ runs inside a Claude Code session, that session's environment variables (`CLAUDECODE`,
69
+ `CLAUDE_CODE_SESSION_ID` and similar) are removed before starting the measured `claude`.
70
+
71
+ `codex exec --json` does not report which model it ran, so pass `--model` if you want the
72
+ model recorded; otherwise the report shows `default`.
73
+
74
+ Other useful flags: `--jobs 4` (parallel runs), `--tasks tag:refactor` or `--tasks a,b`,
75
+ `--keep-workspaces` (inspect what the agent left behind), `--seed` (mock only),
76
+ `--agent-opt bin=/path/to/cli` (run a specific build of the CLI), `--agent-opt effort=high`
77
+ (`claude --effort` or Codex `model_reasoning_effort`).
78
+
79
+ Long runs can be interrupted and continued: name the run with `--run-id` and add `--resume`
80
+ to run only the task and run pairs that `runs.jsonl` does not have yet.
81
+
82
+ ```sh
83
+ rerun-bench run --agent claude --tasks all --runs 3 --out results/ --run-id claude-pilot --yes
84
+ # interrupted; later:
85
+ rerun-bench run --agent claude --tasks all --runs 3 --out results/ --run-id claude-pilot --yes --resume
86
+ ```
87
+
88
+ ## Pilot results
89
+
90
+ A first run against real CLIs on 2026-10-03: all 10 tasks, 3 runs each, Claude Code 2.1.288
91
+ (default model, reported as `claude-opus-5-5`) and Codex CLI 0.160.0 (`gpt-6-luna`), on
92
+ macOS arm64. Full setup, per-task outcomes, raw run records and diffs:
93
+ [docs/pilot-2026-10-03](docs/pilot-2026-10-03/README.md).
94
+
95
+ | Agent / model | Pass rate [Wilson 95% CI] | pass^3 | Flip rate | Median cost/run | Median tokens/run | Median wall time |
96
+ |---|---|---|---|---|---|---|
97
+ | claude / claude-opus-5-5 | 30/30, 100% [89, 100] | 100% | 0% | $0.0886 | 52,017 | 12.6 s |
98
+ | codex / gpt-6-luna | 28/30, 93% [79, 98] | 80% | 13% | not reported | 56,629 | 16.1 s |
99
+
100
+ n = 3 per task is a pilot, not a leaderboard. The pass-rate intervals overlap, so these
101
+ runs do not establish a difference between the two agents. Claude Code's cost is its own
102
+ list-price estimate; Codex reports tokens only.
103
+
104
+ ## Example report
105
+
106
+ Two mock profiles, 10 tasks, 5 runs each (`rerun-bench report results/`):
107
+
108
+ | Agent / model | Runs/task | Pass rate [95% CI] | pass@k | pass^k | Flip rate | Flaky tasks | Cost/run | Cost CV |
109
+ |---|---|---|---|---|---|---|---|---|
110
+ | mock / mock-steady | 5 | 86% [74, 93] | 100% | 40% | 26% | 60% | $0.0601 | 0.19 |
111
+ | mock / mock-flaky | 5 | 60% [46, 72] | 100% | 0% | 50% | 100% | $0.0906 | 0.46 |
112
+
113
+ | Task | mock / mock-steady | mock / mock-flaky |
114
+ |---|---|---|
115
+ | fix-failing-test | `PPPPP` 100%, flip 0%, cost CV 0.13 | `FFFPF` 20%, flip 40%, cost CV 0.49 |
116
+ | minimal-fix | `PPPPF` 80%, flip 40%, cost CV 0.16 | `PPFPF` 60%, flip 60%, cost CV 0.46 |
117
+
118
+ Both profiles reach pass@5 = 100%: given five tries, each solves every task at least once.
119
+ Only pass^5 and the flip rate separate them. The HTML report (`--format html`) is one static
120
+ file with inline CSS and JS, a sortable leaderboard, and a per-task grid of run outcomes.
121
+
122
+ ## Metrics
123
+
124
+ Full definitions, estimators and caveats: [docs/METRICS.md](docs/METRICS.md).
125
+
126
+ | Metric | What it answers |
127
+ |---|---|
128
+ | Pass rate + Wilson 95% CI | How often does a run pass the hidden verifier? |
129
+ | Task-bootstrap 95% CI | Same, with uncertainty over which tasks were sampled (JSON report). |
130
+ | pass@k | Chance that at least one of k runs passes (unbiased estimator). |
131
+ | pass^k | Chance that all k runs pass. The number to watch if you run once and trust the result. |
132
+ | Flip rate | Chance that two runs of the same task disagree, 2c(n-c)/(n(n-1)). |
133
+ | Flaky tasks | Share of tasks with both passes and fails. |
134
+ | Cost / tokens / wall-time CV | Run-to-run spread within a task (std / mean), averaged over tasks. |
135
+ | Cost per success | Total cost divided by passing runs. |
136
+ | Approach similarity | Mean pairwise Jaccard of changed lines among passing runs. 1.0 means the same edit every time. |
137
+
138
+ Only the task's verifier decides pass or fail. The agent's exit code and its own claims of
139
+ success are recorded but not scored.
140
+
141
+ ## Task suite
142
+
143
+ `rerun-bench list` shows the bundled tasks:
144
+
145
+ | Task | What it tests |
146
+ |---|---|
147
+ | `fix-failing-test` | Fix the bug behind a failing unit test without editing the test |
148
+ | `implement-slugify` | Implement a function exactly to a docstring spec |
149
+ | `implement-lru-cache` | Implement a small data structure to spec |
150
+ | `refactor-extract-helper` | Extract duplicated logic; behavior checked on a grid of inputs |
151
+ | `follow-agents-md` | Add a function; the prompt does not mention the repo's AGENTS.md rules, the verifier checks them |
152
+ | `edit-config` | Three precise TOML edits; a production config next to it must stay untouched |
153
+ | `multi-file-rename` | Rename a function across a package, no alias left behind |
154
+ | `minimal-fix` | One-line bug in deliberately dated code; any cleanup outside the function fails |
155
+ | `add-cli-flag` | Add a flag without changing default output |
156
+ | `write-tests` | Write tests that pass on the real code and catch five injected bugs (mutation testing) |
157
+
158
+ Every task is offline, deterministic, and uses only the Python standard library, so it runs
159
+ the same on Linux, macOS and Windows.
160
+
161
+ ## Add a task
162
+
163
+ ```
164
+ tasks/<id>/
165
+ task.toml id, title, prompt, timeout (seconds), tags
166
+ workspace/ the files the agent starts with
167
+ verify.py exit 0 = pass; runs with cwd = the agent's workspace; never shown to the agent
168
+ solution/ reference solution, copied over workspace/ by the test suite
169
+ ```
170
+
171
+ ```toml
172
+ id = "my-task"
173
+ title = "One line describing the task"
174
+ prompt = """
175
+ What you would type to the agent.
176
+ """
177
+ timeout = 600
178
+ tags = ["bugfix", "python"]
179
+ ```
180
+
181
+ Then check it: `rerun-bench --tasks-dir tasks verify-tasks --tasks my-task -v` must print `ok`
182
+ (the untouched workspace fails, the reference solution passes), and `uv run pytest` picks the
183
+ new task up automatically. Rules for verifiers are in [CONTRIBUTING.md](CONTRIBUTING.md).
184
+
185
+ ## Add an adapter
186
+
187
+ Subclass `Adapter` in `src/rerun_bench/adapters/`, implement two pure methods, and register it
188
+ in `ADAPTERS` in `src/rerun_bench/adapters/__init__.py`:
189
+
190
+ ```python
191
+ class MyAgentAdapter(Adapter):
192
+ name = "myagent"
193
+ binary = "myagent"
194
+
195
+ def build_command(self, prompt: str, workspace: Path) -> list[str]:
196
+ return [self.binary, "run", "--json", prompt]
197
+
198
+ def parse_output(self, stdout: str, stderr: str) -> Usage:
199
+ # Fill tokens, cost and model; leave a field None when the CLI does not report it.
200
+ data = json.loads(stdout)
201
+ return Usage(output_tokens=data.get("output_tokens"), cost_usd=data.get("cost"))
202
+ ```
203
+
204
+ The base class handles the subprocess, timeout, wall time and `--version`. Test both methods
205
+ against a captured sample of the CLI's output (see `tests/test_adapters.py`); the test suite
206
+ never calls a real agent.
207
+
208
+ ## Agent skill
209
+
210
+ `skills/rerun-bench/SKILL.md` teaches a coding agent to run the benchmark and add tasks:
211
+
212
+ ```sh
213
+ npx skills add Abelo9996/rerun-bench
214
+ ```
215
+
216
+ ## Roadmap
217
+
218
+ - Public leaderboard, refreshed on model launch days, built from the HTML report.
219
+ - Version-over-version tracking: the same model under successive CLI releases, with
220
+ per-task significance tests.
221
+ - Paired comparisons between two result sets (Fisher exact per task, task-level bootstrap
222
+ for the suite).
223
+ - More tasks in other languages, kept small, offline and deterministic.
224
+
225
+ ## Related projects
226
+
227
+ - [nerf-watch](https://github.com/Abelo9996/nerf-watch): detects silent model and cost changes
228
+ from your local agent logs.
229
+ - [snap-back](https://github.com/Abelo9996/snap-back): undo for any coding agent.
230
+
231
+ ## Development
232
+
233
+ ```sh
234
+ uv sync
235
+ uv run pytest
236
+ uv run ruff check . && uv run ruff format --check .
237
+ uv run rerun-bench verify-tasks
238
+ ```
239
+
240
+ MIT licensed. See [LICENSE](LICENSE).