agent-testbench 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. agent_testbench-0.1.0/.gitignore +10 -0
  2. agent_testbench-0.1.0/LICENSE +21 -0
  3. agent_testbench-0.1.0/PKG-INFO +301 -0
  4. agent_testbench-0.1.0/README.md +262 -0
  5. agent_testbench-0.1.0/examples/tiny-model/README.md +26 -0
  6. agent_testbench-0.1.0/examples/web-ui/README.md +7 -0
  7. agent_testbench-0.1.0/pyproject.toml +57 -0
  8. agent_testbench-0.1.0/src/agent_testbench/__init__.py +3 -0
  9. agent_testbench-0.1.0/src/agent_testbench/__main__.py +3 -0
  10. agent_testbench-0.1.0/src/agent_testbench/backends/__init__.py +59 -0
  11. agent_testbench-0.1.0/src/agent_testbench/backends/colab_backend.py +157 -0
  12. agent_testbench-0.1.0/src/agent_testbench/backends/local.py +25 -0
  13. agent_testbench-0.1.0/src/agent_testbench/backends/modal_backend.py +212 -0
  14. agent_testbench-0.1.0/src/agent_testbench/budget.py +113 -0
  15. agent_testbench-0.1.0/src/agent_testbench/cli.py +514 -0
  16. agent_testbench-0.1.0/src/agent_testbench/colab_shim/README.md +5 -0
  17. agent_testbench-0.1.0/src/agent_testbench/colab_shim/google/colab/__init__.py +3 -0
  18. agent_testbench-0.1.0/src/agent_testbench/colab_shim/google/colab/drive.py +11 -0
  19. agent_testbench-0.1.0/src/agent_testbench/colab_shim/google/colab/files.py +45 -0
  20. agent_testbench-0.1.0/src/agent_testbench/colab_shim/google/colab/output.py +25 -0
  21. agent_testbench-0.1.0/src/agent_testbench/colab_shim/google/colab/patches.py +10 -0
  22. agent_testbench-0.1.0/src/agent_testbench/colab_shim/google/colab/userdata.py +23 -0
  23. agent_testbench-0.1.0/src/agent_testbench/config.py +188 -0
  24. agent_testbench-0.1.0/src/agent_testbench/driver.py +72 -0
  25. agent_testbench-0.1.0/src/agent_testbench/executor.py +255 -0
  26. agent_testbench-0.1.0/src/agent_testbench/expect.py +156 -0
  27. agent_testbench-0.1.0/src/agent_testbench/kernelkit/__init__.py +3 -0
  28. agent_testbench-0.1.0/src/agent_testbench/kernelkit/client.py +138 -0
  29. agent_testbench-0.1.0/src/agent_testbench/kernelkit/launcher.py +61 -0
  30. agent_testbench-0.1.0/src/agent_testbench/lint.py +82 -0
  31. agent_testbench-0.1.0/src/agent_testbench/mcp_server.py +169 -0
  32. agent_testbench-0.1.0/src/agent_testbench/mocks.py +74 -0
  33. agent_testbench-0.1.0/src/agent_testbench/notebook.py +274 -0
  34. agent_testbench-0.1.0/src/agent_testbench/report.py +129 -0
  35. agent_testbench-0.1.0/src/agent_testbench/runs.py +236 -0
  36. agent_testbench-0.1.0/src/agent_testbench/session_backends.py +471 -0
  37. agent_testbench-0.1.0/src/agent_testbench/sessions.py +298 -0
  38. agent_testbench-0.1.0/src/agent_testbench/web.py +129 -0
@@ -0,0 +1,10 @@
1
+ .venv/
2
+ __pycache__/
3
+ *.pyc
4
+ *.egg-info/
5
+ dist/
6
+ build/
7
+ .pytest_cache/
8
+ .testbench/
9
+ examples/*/outputs/
10
+ .DS_Store
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Sarp Tandoven
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,301 @@
1
+ Metadata-Version: 2.5
2
+ Name: agent-testbench
3
+ Version: 0.1.0
4
+ Summary: Let a coding agent iterate on notebooks, models and UIs like an testbench: run, read the evidence, fix, rerun - locally, on Modal or on Colab, inside a budget.
5
+ Project-URL: Homepage, https://github.com/sarptandoven/agent-testbench
6
+ Project-URL: Issues, https://github.com/sarptandoven/agent-testbench/issues
7
+ Author: Sarp Tandoven
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: agents,claude,claude-code,colab,gpu,mcp,modal,notebooks,testing
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
16
+ Classifier: Topic :: Software Development :: Testing
17
+ Requires-Python: >=3.10
18
+ Requires-Dist: ipykernel>=6.29
19
+ Requires-Dist: jupyter-client>=8
20
+ Requires-Dist: nbclient>=0.10
21
+ Requires-Dist: nbformat>=5.9
22
+ Requires-Dist: pyyaml>=6
23
+ Provides-Extra: all
24
+ Requires-Dist: mcp>=1.2; extra == 'all'
25
+ Requires-Dist: modal>=1.5; extra == 'all'
26
+ Requires-Dist: playwright>=1.45; extra == 'all'
27
+ Provides-Extra: dev
28
+ Requires-Dist: mcp>=1.2; extra == 'dev'
29
+ Requires-Dist: modal>=1.5; extra == 'dev'
30
+ Requires-Dist: playwright>=1.45; extra == 'dev'
31
+ Requires-Dist: pytest>=8; extra == 'dev'
32
+ Provides-Extra: mcp
33
+ Requires-Dist: mcp>=1.2; extra == 'mcp'
34
+ Provides-Extra: modal
35
+ Requires-Dist: modal>=1.5; extra == 'modal'
36
+ Provides-Extra: web
37
+ Requires-Dist: playwright>=1.45; extra == 'web'
38
+ Description-Content-Type: text/markdown
39
+
40
+ # agent-testbench
41
+
42
+ [![tests](https://github.com/sarptandoven/agent-testbench/actions/workflows/ci.yml/badge.svg)](https://github.com/sarptandoven/agent-testbench/actions/workflows/ci.yml)
43
+ [![PyPI](https://img.shields.io/pypi/v/agent-testbench.svg)](https://pypi.org/project/agent-testbench/)
44
+ [![license: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://github.com/sarptandoven/agent-testbench/blob/main/LICENSE)
45
+
46
+ **A test bench your coding agent operates.** Claude, GPT/Codex or any agent can start a live kernel on a Modal GPU,
47
+ a Colab VM or your machine, run code and notebook cells in it, inspect variables and plots, install packages,
48
+ move files, fix and rerun - and then prove the result with a clean, checked run. Everything stays inside a
49
+ budget you set.
50
+
51
+ ```
52
+ $ testbench session start gpu --backend modal --gpu T4 --idle-minutes 3 --max-minutes 8
53
+ gpu ready on modal T4 up 0.0 min, idle 0.0 min (stops after 3 idle / 8 total) ~$0.000 so far
54
+ Within budget: up to $0.16 (today so far $0.01 of $2.00).
55
+
56
+ $ testbench exec -f check_gpu.py
57
+ [check_gpu.py | ok | 3.3 s]
58
+ 20 matmuls of 4096x4096 on Tesla T4: 0.67 s
59
+ => torch.Size([4096, 4096])
60
+
61
+ $ testbench exec "float(b.mean()), torch.cuda.max_memory_allocated() // 2**20" # variables persist
62
+ [code | ok | 0.2 s]
63
+ => (1023.1211547851562, 200)
64
+
65
+ $ testbench session stop
66
+ gpu stopped after 1.4 min, about $0.017
67
+ ```
68
+
69
+ ## What your agent can do with it
70
+
71
+ - **Work live** in a kernel that stays up - on this machine, in a [Modal](https://modal.com) Sandbox with any
72
+ GPU (the machinery behind Modal Notebooks), or on a [Colab](https://colab.research.google.com) VM through
73
+ Google's Colab CLI. Run code, `.py` files, or a notebook's cells with its Colab form fields set; get back what
74
+ it printed, the last value, plots as image files, and errors with the line that raised them. Install
75
+ packages, run shell commands (`nvidia-smi`), list, upload and download files, and push local edits with
76
+ `sync` without losing variables.
77
+ - **Prove with plans**: run notebooks (Colab ones included), scripts or web UIs end to end on any of the same
78
+ backends, checked against expectations - metric thresholds, files, video frame counts, text on a page - with
79
+ a report that names the failing cell and line, shows the evidence, and says what changed since the last run.
80
+ - **Stay inside a budget**: every session and run is priced before it starts (worst case = its time limit on
81
+ its hardware), refused above your cap, sent to you for approval above your threshold, and stopped by the
82
+ provider at its limit even if your laptop goes to sleep. Sessions stop themselves when idle.
83
+ - **Stay safe**: secrets are referred to by name only, paid APIs get strict mocks in tests, and the Claude Code
84
+ hook blocks raw cloud launches and commands that would print secrets.
85
+
86
+ ## Install
87
+
88
+ ```bash
89
+ pip install "agent-testbench[all]" # or: uv tool install "agent-testbench[all]"
90
+ ```
91
+
92
+ The latest from GitHub: `pip install "agent-testbench[all] @ git+https://github.com/sarptandoven/agent-testbench"`.
93
+
94
+ - **Modal**: `modal setup` once (your account, your card; [pricing](https://modal.com/pricing)).
95
+ - **Colab**: `uv tool install google-colab-cli`, then log in once (`colab new` opens the sign-in; or set
96
+ `colab: {auth: adc}` if you use gcloud Application Default Credentials). Uses your Colab compute units.
97
+ - **Web UI checks**: `playwright install chromium`.
98
+
99
+ `testbench doctor` shows what this machine can do. Then, in your project:
100
+
101
+ ```bash
102
+ testbench init # writes testbench.yaml (budget, session defaults, a plan per notebook) and AGENTS.md
103
+ ```
104
+
105
+ ## Connect your agent
106
+
107
+ **Claude Code** - the plugin adds the skills and the guard hook:
108
+
109
+ ```bash
110
+ claude plugin marketplace add sarptandoven/agent-testbench
111
+ claude plugin install agent-testbench@agent-testbench
112
+ ```
113
+
114
+ Then ask in plain words ("get training working on a T4 and prove it with the smoke plan"), or call the skills:
115
+ `/agent-testbench:iterate` and `/agent-testbench:setup`.
116
+
117
+ **Claude Desktop, Cursor, Codex and other MCP clients**:
118
+
119
+ ```json
120
+ { "mcpServers": { "agent-testbench": { "command": "testbench", "args": ["mcp"] } } }
121
+ ```
122
+
123
+ Tools: `session_start`, `session_exec`, `session_files`, `session_sync`, `session_install`, `session_shell`,
124
+ `session_list`, `session_stop`, `plans`, `run_plan`, `wait_run`, `get_report`, `list_runs`, `cancel_run`,
125
+ `spend`, `lint_notebook`, `add_note`. In Claude Code: `claude mcp add agent-testbench -- testbench mcp`.
126
+
127
+ **Any other agent** reads the rules `testbench init` writes into `AGENTS.md`.
128
+
129
+ ## Live sessions
130
+
131
+ ```bash
132
+ testbench session start # default from testbench.yaml (local unless set)
133
+ testbench session start big --backend modal --gpu A100-80GB --idle-minutes 10 --max-minutes 90
134
+ testbench session start nb --backend colab --gpu T4
135
+
136
+ testbench exec "x = load_data(); x.shape" # code
137
+ testbench exec -f debug/check.py # a local file
138
+ testbench exec --cells train.ipynb:1-4 --params '{"EPOCHS": 1}' # notebook cells, form fields set
139
+ testbench files get outputs/sample.png # ls / get / put on the session's machine
140
+ testbench session sync # copy local edits again; variables stay
141
+ testbench session install "transformers==4.46.3"
142
+ testbench session shell "nvidia-smi"
143
+ testbench session list # uptime, idle time, cost so far
144
+ testbench session stop # or let it stop itself
145
+ ```
146
+
147
+ | backend | where the kernel runs | stops |
148
+ |---|---|---|
149
+ | `local` | this machine, in the project folder | after `idle_minutes` unused or `max_minutes` |
150
+ | `modal` | a Modal Sandbox with your GPU (T4 ... H100, B200), project in `/work/project` and `/content`, a persistent volume at `/testbench` | Modal enforces both limits itself (a running command counts as activity), so it stops even if your laptop is off |
151
+ | `colab` | a Colab VM (CPU, T4, L4, A100, H100 by subscription), project in `/content` | after `idle_minutes` unused or `max_minutes` (Colab may also reclaim VMs after an hour or two) |
152
+
153
+ A call that runs past `--timeout` is interrupted; the kernel and its variables survive. One command runs at a
154
+ time per session.
155
+
156
+ **GPU libraries on Modal**: the default image is Debian slim with the NVIDIA driver but no CUDA toolkit.
157
+ PyTorch works as it is (its wheels bring their own CUDA libraries - checked on a T4). Libraries that expect the
158
+ CUDA toolkit on the machine, such as CuPy, fail there with errors like
159
+ `libcurand.so.10: cannot open shared object file` (also checked). Give them the CUDA libraries they need as
160
+ described in their install guide (for CuPy: docs.cupy.dev, "Installation"), and confirm it in a short GPU
161
+ session (`testbench session start --backend modal --gpu T4 --max-minutes 5`) before a long run.
162
+
163
+ Notebooks see a headless `google.colab` (uploads from files you name, downloads kept,
164
+ secrets from the environment, Drive as a local folder), so Colab notebooks run as they are.
165
+
166
+ ## Plans
167
+
168
+ ```yaml
169
+ project: my-model # names the Modal app and volume
170
+
171
+ budget:
172
+ max_run_usd: 2.00 # refuse any run or session that could cost more
173
+ ask_above_usd: 0.50 # above this a person approves (--approve)
174
+ max_day_usd: 10.00 # everything today together
175
+
176
+ sessions:
177
+ backend: modal
178
+ gpu: A10
179
+ idle_minutes: 10
180
+ max_minutes: 60
181
+
182
+ modal: # image for Modal plans and sessions
183
+ python: "3.11"
184
+ pip: [torch==2.5.1, scikit-learn==1.5.2]
185
+ cpu: 4
186
+ memory_gb: 16
187
+
188
+ plans:
189
+ smoke: # cheapest proof first
190
+ backend: local
191
+ steps:
192
+ - name: lint
193
+ run: testbench lint train.ipynb --strict
194
+ max_minutes: 1
195
+ - name: train tiny
196
+ notebook: train.ipynb
197
+ params: {EPOCHS: 1, SUBSET: 200} # Colab form fields (NAME = value #@param)
198
+ max_minutes: 5
199
+ expect:
200
+ json: {outputs/metrics.json: {loss: "< 2.0"}}
201
+
202
+ gpu-check: # a few minutes on the target GPU before a long run
203
+ backend: modal
204
+ gpu: A10
205
+ steps:
206
+ - name: train short
207
+ notebook: train.ipynb
208
+ params: {EPOCHS: 2}
209
+ max_minutes: 10 # also the hard stop on Modal's side
210
+ artifacts: [outputs/*.png]
211
+ expect:
212
+ json: {outputs/metrics.json: {accuracy: ">= 0.8"}}
213
+ ```
214
+
215
+ ```bash
216
+ testbench plans # every plan, its hardware and worst-case cost
217
+ testbench run smoke # runs it and prints the report (follows up to 9 min; then `testbench wait`)
218
+ testbench report # the latest report
219
+ ```
220
+
221
+ A report leads with what failed: the cell, the error, the line that raised it, each expectation that missed and
222
+ what was found instead, then what changed since the previous run of the plan ("Fixed: train", "Broke: eval",
223
+ "accuracy 0.71 -> 0.93"), the evidence files to look at, and GPU use per step (an idle GPU shows as 0%).
224
+ Steps are `notebook:`, `run:` (a shell command) or `web:` (a page driven in headless Chromium: click, fill,
225
+ press, wait for, screenshot). The full reference is in [docs/reference.md](https://github.com/sarptandoven/agent-testbench/blob/main/docs/reference.md).
226
+
227
+ ## Guardrails
228
+
229
+ - **Budget**: worst case = hardware price x time limit (+ start-up). Over `max_run_usd` or the day's cap:
230
+ refused. Over `ask_above_usd`: needs `--approve`, which the agent is told never to pass itself - and in Claude
231
+ Code the hook turns it into a question to you. Every finished run and session is written to a ledger;
232
+ `testbench spend --modal` also reads Modal's own bill.
233
+ - **Hard stops**: plans and sessions on Modal carry their limits in Modal itself (function and Sandbox
234
+ timeouts, no retries, 2-second scale-down). Colab sessions and runs are stopped by their supervisor and always
235
+ released, even when a run fails.
236
+ - **Secrets**: by name only; values come from Modal secrets or your environment. The hook refuses commands that
237
+ would print them.
238
+ - **Paid APIs**: test with strict mocks (`agent_testbench.mocks.StrictAPI`): a module with the real client's
239
+ name that checks arguments against the published schema and refuses anything unexpected.
240
+ - **Outside world**: the hook asks before `git push`, `gh repo create` and anything in `guard.ask`.
241
+
242
+ The hook is a safety net for agents, not a security boundary; the budget checks and remote time limits are
243
+ enforced in code either way.
244
+
245
+ ## Notebooks
246
+
247
+ - `testbench nb split nb.ipynb cells/` and `testbench nb build cells/ nb.ipynb` - edit cells as plain files.
248
+ - `testbench nb fields nb.ipynb` - the form fields a plan or `--params` can set.
249
+ - `testbench lint nb.ipynb` - big saved outputs, embedded video, secrets without a fallback, `files.upload()`,
250
+ unpinned installs, the inline matplotlib backend leaking into child processes, broken form fields, hard-coded
251
+ `/content` paths.
252
+
253
+ ## Commands
254
+
255
+ | | |
256
+ |---|---|
257
+ | `testbench session start/list/stop/sync/install/shell` | live kernels |
258
+ | `testbench exec` / `testbench files` | work in a session |
259
+ | `testbench plans` / `run` / `wait` / `report` / `runs` / `cancel` | plans |
260
+ | `testbench estimate <plan>` / `spend [--modal]` | money |
261
+ | `testbench lint` / `nb split/build/fields` | notebooks |
262
+ | `testbench note "..."` | the project's lab notebook (`.testbench/NOTES.md`) |
263
+ | `testbench init` / `doctor` / `mcp` | set up, check, serve MCP |
264
+
265
+ ## Examples
266
+
267
+ - [`examples/tiny-model`](https://github.com/sarptandoven/agent-testbench/tree/main/examples/tiny-model): a Colab-style notebook that trains a small network, with plans
268
+ for local, Modal CPU, Modal T4 and Colab.
269
+ [Open in Colab](https://colab.research.google.com/github/sarptandoven/agent-testbench/blob/main/examples/tiny-model/tiny_model.ipynb)
270
+ - [`examples/web-ui`](https://github.com/sarptandoven/agent-testbench/tree/main/examples/web-ui): a page checked in a headless browser - clicks, text, console errors, screenshots.
271
+
272
+ ## How it is tested
273
+
274
+ `python -m pytest` runs everything that needs no account: config, notebooks executed in real kernels (errors,
275
+ timeouts, images, magics, the Colab stand-in), expectations, budgets, lint, mocks, the generated Modal app, full
276
+ CLI runs (pass, fail, fix, compare, cancel, timeouts, refusals), live sessions on this machine (persistent
277
+ state, interrupts, notebook cells, files, auto-stop, budgets), the guard hook, the MCP server over stdio, and web
278
+ steps in headless Chromium. It passes on Python 3.11 and 3.12.
279
+
280
+ `tests/test_live.py` runs the same things on the real services and is opt-in because it costs money:
281
+ `TESTBENCH_TEST_MODAL=1` (plans on CPU and a T4, cancelling a GPU run mid-flight, a live session, a T4 session)
282
+ and `TESTBENCH_TEST_COLAB=1` (a plan and a live session on a Colab CPU VM). Colab T4 sessions and plans were also
283
+ run by hand: PyTorch and CuPy on the GPU, a 70-second cell, an interrupted cell whose kernel kept its state, plots
284
+ coming back as images.
285
+
286
+ [docs/demo.md](https://github.com/sarptandoven/agent-testbench/blob/main/docs/demo.md) shows Claude Code, given only this plugin and a project with a planted bug, finding
287
+ and fixing it and proving the fix with a run.
288
+
289
+ ## Limits
290
+
291
+ - Costs are estimates (time x published price); `testbench spend --modal` reads Modal's actual bill. Colab
292
+ unit rates vary by account - put yours (`colab usage`) under `pricing: colab_units_per_hour:`.
293
+ - It runs `.ipynb` files and live kernels; it does not drive the Colab or Modal Notebooks web pages themselves
294
+ (Modal Notebooks have no API; for steering a Colab tab in your browser see Google's
295
+ [colab-mcp](https://github.com/googlecolab/colab-mcp)).
296
+ - Colab VMs can be reclaimed after an hour or two even while busy; use Modal for long jobs.
297
+ - macOS and Linux. Windows is untested (the Colab CLI does not support it either).
298
+
299
+ ## License
300
+
301
+ MIT
@@ -0,0 +1,262 @@
1
+ # agent-testbench
2
+
3
+ [![tests](https://github.com/sarptandoven/agent-testbench/actions/workflows/ci.yml/badge.svg)](https://github.com/sarptandoven/agent-testbench/actions/workflows/ci.yml)
4
+ [![PyPI](https://img.shields.io/pypi/v/agent-testbench.svg)](https://pypi.org/project/agent-testbench/)
5
+ [![license: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://github.com/sarptandoven/agent-testbench/blob/main/LICENSE)
6
+
7
+ **A test bench your coding agent operates.** Claude, GPT/Codex or any agent can start a live kernel on a Modal GPU,
8
+ a Colab VM or your machine, run code and notebook cells in it, inspect variables and plots, install packages,
9
+ move files, fix and rerun - and then prove the result with a clean, checked run. Everything stays inside a
10
+ budget you set.
11
+
12
+ ```
13
+ $ testbench session start gpu --backend modal --gpu T4 --idle-minutes 3 --max-minutes 8
14
+ gpu ready on modal T4 up 0.0 min, idle 0.0 min (stops after 3 idle / 8 total) ~$0.000 so far
15
+ Within budget: up to $0.16 (today so far $0.01 of $2.00).
16
+
17
+ $ testbench exec -f check_gpu.py
18
+ [check_gpu.py | ok | 3.3 s]
19
+ 20 matmuls of 4096x4096 on Tesla T4: 0.67 s
20
+ => torch.Size([4096, 4096])
21
+
22
+ $ testbench exec "float(b.mean()), torch.cuda.max_memory_allocated() // 2**20" # variables persist
23
+ [code | ok | 0.2 s]
24
+ => (1023.1211547851562, 200)
25
+
26
+ $ testbench session stop
27
+ gpu stopped after 1.4 min, about $0.017
28
+ ```
29
+
30
+ ## What your agent can do with it
31
+
32
+ - **Work live** in a kernel that stays up - on this machine, in a [Modal](https://modal.com) Sandbox with any
33
+ GPU (the machinery behind Modal Notebooks), or on a [Colab](https://colab.research.google.com) VM through
34
+ Google's Colab CLI. Run code, `.py` files, or a notebook's cells with its Colab form fields set; get back what
35
+ it printed, the last value, plots as image files, and errors with the line that raised them. Install
36
+ packages, run shell commands (`nvidia-smi`), list, upload and download files, and push local edits with
37
+ `sync` without losing variables.
38
+ - **Prove with plans**: run notebooks (Colab ones included), scripts or web UIs end to end on any of the same
39
+ backends, checked against expectations - metric thresholds, files, video frame counts, text on a page - with
40
+ a report that names the failing cell and line, shows the evidence, and says what changed since the last run.
41
+ - **Stay inside a budget**: every session and run is priced before it starts (worst case = its time limit on
42
+ its hardware), refused above your cap, sent to you for approval above your threshold, and stopped by the
43
+ provider at its limit even if your laptop goes to sleep. Sessions stop themselves when idle.
44
+ - **Stay safe**: secrets are referred to by name only, paid APIs get strict mocks in tests, and the Claude Code
45
+ hook blocks raw cloud launches and commands that would print secrets.
46
+
47
+ ## Install
48
+
49
+ ```bash
50
+ pip install "agent-testbench[all]" # or: uv tool install "agent-testbench[all]"
51
+ ```
52
+
53
+ The latest from GitHub: `pip install "agent-testbench[all] @ git+https://github.com/sarptandoven/agent-testbench"`.
54
+
55
+ - **Modal**: `modal setup` once (your account, your card; [pricing](https://modal.com/pricing)).
56
+ - **Colab**: `uv tool install google-colab-cli`, then log in once (`colab new` opens the sign-in; or set
57
+ `colab: {auth: adc}` if you use gcloud Application Default Credentials). Uses your Colab compute units.
58
+ - **Web UI checks**: `playwright install chromium`.
59
+
60
+ `testbench doctor` shows what this machine can do. Then, in your project:
61
+
62
+ ```bash
63
+ testbench init # writes testbench.yaml (budget, session defaults, a plan per notebook) and AGENTS.md
64
+ ```
65
+
66
+ ## Connect your agent
67
+
68
+ **Claude Code** - the plugin adds the skills and the guard hook:
69
+
70
+ ```bash
71
+ claude plugin marketplace add sarptandoven/agent-testbench
72
+ claude plugin install agent-testbench@agent-testbench
73
+ ```
74
+
75
+ Then ask in plain words ("get training working on a T4 and prove it with the smoke plan"), or call the skills:
76
+ `/agent-testbench:iterate` and `/agent-testbench:setup`.
77
+
78
+ **Claude Desktop, Cursor, Codex and other MCP clients**:
79
+
80
+ ```json
81
+ { "mcpServers": { "agent-testbench": { "command": "testbench", "args": ["mcp"] } } }
82
+ ```
83
+
84
+ Tools: `session_start`, `session_exec`, `session_files`, `session_sync`, `session_install`, `session_shell`,
85
+ `session_list`, `session_stop`, `plans`, `run_plan`, `wait_run`, `get_report`, `list_runs`, `cancel_run`,
86
+ `spend`, `lint_notebook`, `add_note`. In Claude Code: `claude mcp add agent-testbench -- testbench mcp`.
87
+
88
+ **Any other agent** reads the rules `testbench init` writes into `AGENTS.md`.
89
+
90
+ ## Live sessions
91
+
92
+ ```bash
93
+ testbench session start # default from testbench.yaml (local unless set)
94
+ testbench session start big --backend modal --gpu A100-80GB --idle-minutes 10 --max-minutes 90
95
+ testbench session start nb --backend colab --gpu T4
96
+
97
+ testbench exec "x = load_data(); x.shape" # code
98
+ testbench exec -f debug/check.py # a local file
99
+ testbench exec --cells train.ipynb:1-4 --params '{"EPOCHS": 1}' # notebook cells, form fields set
100
+ testbench files get outputs/sample.png # ls / get / put on the session's machine
101
+ testbench session sync # copy local edits again; variables stay
102
+ testbench session install "transformers==4.46.3"
103
+ testbench session shell "nvidia-smi"
104
+ testbench session list # uptime, idle time, cost so far
105
+ testbench session stop # or let it stop itself
106
+ ```
107
+
108
+ | backend | where the kernel runs | stops |
109
+ |---|---|---|
110
+ | `local` | this machine, in the project folder | after `idle_minutes` unused or `max_minutes` |
111
+ | `modal` | a Modal Sandbox with your GPU (T4 ... H100, B200), project in `/work/project` and `/content`, a persistent volume at `/testbench` | Modal enforces both limits itself (a running command counts as activity), so it stops even if your laptop is off |
112
+ | `colab` | a Colab VM (CPU, T4, L4, A100, H100 by subscription), project in `/content` | after `idle_minutes` unused or `max_minutes` (Colab may also reclaim VMs after an hour or two) |
113
+
114
+ A call that runs past `--timeout` is interrupted; the kernel and its variables survive. One command runs at a
115
+ time per session.
116
+
117
+ **GPU libraries on Modal**: the default image is Debian slim with the NVIDIA driver but no CUDA toolkit.
118
+ PyTorch works as it is (its wheels bring their own CUDA libraries - checked on a T4). Libraries that expect the
119
+ CUDA toolkit on the machine, such as CuPy, fail there with errors like
120
+ `libcurand.so.10: cannot open shared object file` (also checked). Give them the CUDA libraries they need as
121
+ described in their install guide (for CuPy: docs.cupy.dev, "Installation"), and confirm it in a short GPU
122
+ session (`testbench session start --backend modal --gpu T4 --max-minutes 5`) before a long run.
123
+
124
+ Notebooks see a headless `google.colab` (uploads from files you name, downloads kept,
125
+ secrets from the environment, Drive as a local folder), so Colab notebooks run as they are.
126
+
127
+ ## Plans
128
+
129
+ ```yaml
130
+ project: my-model # names the Modal app and volume
131
+
132
+ budget:
133
+ max_run_usd: 2.00 # refuse any run or session that could cost more
134
+ ask_above_usd: 0.50 # above this a person approves (--approve)
135
+ max_day_usd: 10.00 # everything today together
136
+
137
+ sessions:
138
+ backend: modal
139
+ gpu: A10
140
+ idle_minutes: 10
141
+ max_minutes: 60
142
+
143
+ modal: # image for Modal plans and sessions
144
+ python: "3.11"
145
+ pip: [torch==2.5.1, scikit-learn==1.5.2]
146
+ cpu: 4
147
+ memory_gb: 16
148
+
149
+ plans:
150
+ smoke: # cheapest proof first
151
+ backend: local
152
+ steps:
153
+ - name: lint
154
+ run: testbench lint train.ipynb --strict
155
+ max_minutes: 1
156
+ - name: train tiny
157
+ notebook: train.ipynb
158
+ params: {EPOCHS: 1, SUBSET: 200} # Colab form fields (NAME = value #@param)
159
+ max_minutes: 5
160
+ expect:
161
+ json: {outputs/metrics.json: {loss: "< 2.0"}}
162
+
163
+ gpu-check: # a few minutes on the target GPU before a long run
164
+ backend: modal
165
+ gpu: A10
166
+ steps:
167
+ - name: train short
168
+ notebook: train.ipynb
169
+ params: {EPOCHS: 2}
170
+ max_minutes: 10 # also the hard stop on Modal's side
171
+ artifacts: [outputs/*.png]
172
+ expect:
173
+ json: {outputs/metrics.json: {accuracy: ">= 0.8"}}
174
+ ```
175
+
176
+ ```bash
177
+ testbench plans # every plan, its hardware and worst-case cost
178
+ testbench run smoke # runs it and prints the report (follows up to 9 min; then `testbench wait`)
179
+ testbench report # the latest report
180
+ ```
181
+
182
+ A report leads with what failed: the cell, the error, the line that raised it, each expectation that missed and
183
+ what was found instead, then what changed since the previous run of the plan ("Fixed: train", "Broke: eval",
184
+ "accuracy 0.71 -> 0.93"), the evidence files to look at, and GPU use per step (an idle GPU shows as 0%).
185
+ Steps are `notebook:`, `run:` (a shell command) or `web:` (a page driven in headless Chromium: click, fill,
186
+ press, wait for, screenshot). The full reference is in [docs/reference.md](https://github.com/sarptandoven/agent-testbench/blob/main/docs/reference.md).
187
+
188
+ ## Guardrails
189
+
190
+ - **Budget**: worst case = hardware price x time limit (+ start-up). Over `max_run_usd` or the day's cap:
191
+ refused. Over `ask_above_usd`: needs `--approve`, which the agent is told never to pass itself - and in Claude
192
+ Code the hook turns it into a question to you. Every finished run and session is written to a ledger;
193
+ `testbench spend --modal` also reads Modal's own bill.
194
+ - **Hard stops**: plans and sessions on Modal carry their limits in Modal itself (function and Sandbox
195
+ timeouts, no retries, 2-second scale-down). Colab sessions and runs are stopped by their supervisor and always
196
+ released, even when a run fails.
197
+ - **Secrets**: by name only; values come from Modal secrets or your environment. The hook refuses commands that
198
+ would print them.
199
+ - **Paid APIs**: test with strict mocks (`agent_testbench.mocks.StrictAPI`): a module with the real client's
200
+ name that checks arguments against the published schema and refuses anything unexpected.
201
+ - **Outside world**: the hook asks before `git push`, `gh repo create` and anything in `guard.ask`.
202
+
203
+ The hook is a safety net for agents, not a security boundary; the budget checks and remote time limits are
204
+ enforced in code either way.
205
+
206
+ ## Notebooks
207
+
208
+ - `testbench nb split nb.ipynb cells/` and `testbench nb build cells/ nb.ipynb` - edit cells as plain files.
209
+ - `testbench nb fields nb.ipynb` - the form fields a plan or `--params` can set.
210
+ - `testbench lint nb.ipynb` - big saved outputs, embedded video, secrets without a fallback, `files.upload()`,
211
+ unpinned installs, the inline matplotlib backend leaking into child processes, broken form fields, hard-coded
212
+ `/content` paths.
213
+
214
+ ## Commands
215
+
216
+ | | |
217
+ |---|---|
218
+ | `testbench session start/list/stop/sync/install/shell` | live kernels |
219
+ | `testbench exec` / `testbench files` | work in a session |
220
+ | `testbench plans` / `run` / `wait` / `report` / `runs` / `cancel` | plans |
221
+ | `testbench estimate <plan>` / `spend [--modal]` | money |
222
+ | `testbench lint` / `nb split/build/fields` | notebooks |
223
+ | `testbench note "..."` | the project's lab notebook (`.testbench/NOTES.md`) |
224
+ | `testbench init` / `doctor` / `mcp` | set up, check, serve MCP |
225
+
226
+ ## Examples
227
+
228
+ - [`examples/tiny-model`](https://github.com/sarptandoven/agent-testbench/tree/main/examples/tiny-model): a Colab-style notebook that trains a small network, with plans
229
+ for local, Modal CPU, Modal T4 and Colab.
230
+ [Open in Colab](https://colab.research.google.com/github/sarptandoven/agent-testbench/blob/main/examples/tiny-model/tiny_model.ipynb)
231
+ - [`examples/web-ui`](https://github.com/sarptandoven/agent-testbench/tree/main/examples/web-ui): a page checked in a headless browser - clicks, text, console errors, screenshots.
232
+
233
+ ## How it is tested
234
+
235
+ `python -m pytest` runs everything that needs no account: config, notebooks executed in real kernels (errors,
236
+ timeouts, images, magics, the Colab stand-in), expectations, budgets, lint, mocks, the generated Modal app, full
237
+ CLI runs (pass, fail, fix, compare, cancel, timeouts, refusals), live sessions on this machine (persistent
238
+ state, interrupts, notebook cells, files, auto-stop, budgets), the guard hook, the MCP server over stdio, and web
239
+ steps in headless Chromium. It passes on Python 3.11 and 3.12.
240
+
241
+ `tests/test_live.py` runs the same things on the real services and is opt-in because it costs money:
242
+ `TESTBENCH_TEST_MODAL=1` (plans on CPU and a T4, cancelling a GPU run mid-flight, a live session, a T4 session)
243
+ and `TESTBENCH_TEST_COLAB=1` (a plan and a live session on a Colab CPU VM). Colab T4 sessions and plans were also
244
+ run by hand: PyTorch and CuPy on the GPU, a 70-second cell, an interrupted cell whose kernel kept its state, plots
245
+ coming back as images.
246
+
247
+ [docs/demo.md](https://github.com/sarptandoven/agent-testbench/blob/main/docs/demo.md) shows Claude Code, given only this plugin and a project with a planted bug, finding
248
+ and fixing it and proving the fix with a run.
249
+
250
+ ## Limits
251
+
252
+ - Costs are estimates (time x published price); `testbench spend --modal` reads Modal's actual bill. Colab
253
+ unit rates vary by account - put yours (`colab usage`) under `pricing: colab_units_per_hour:`.
254
+ - It runs `.ipynb` files and live kernels; it does not drive the Colab or Modal Notebooks web pages themselves
255
+ (Modal Notebooks have no API; for steering a Colab tab in your browser see Google's
256
+ [colab-mcp](https://github.com/googlecolab/colab-mcp)).
257
+ - Colab VMs can be reclaimed after an hour or two even while busy; use Modal for long jobs.
258
+ - macOS and Linux. Windows is untested (the Colab CLI does not support it either).
259
+
260
+ ## License
261
+
262
+ MIT
@@ -0,0 +1,26 @@
1
+ # tiny-model
2
+
3
+ A Colab-style notebook (`tiny_model.ipynb`, built from `cells/`) that trains a small neural network on
4
+ scikit-learn's bundled digits, saves `outputs/metrics.json` and a confusion-matrix plot, and offers the
5
+ metrics for download. Its form fields (`HIDDEN`, `EPOCHS`, `LEARNING_RATE`, `SEED`) are set per plan.
6
+
7
+ ```bash
8
+ pip install "agent-testbench[all]" scikit-learn==1.5.2 matplotlib==3.9.2
9
+ testbench plans
10
+ testbench run smoke # lint + quick train here, free, seconds
11
+ testbench run full-modal # full train in a Modal CPU container (~$0.002)
12
+ testbench run gpu-check # nvidia-smi + train on a Modal T4 (~$0.013) - the report shows the GPU sat idle
13
+ testbench run colab-cpu # same notebook on a Colab CPU runtime (needs the Colab CLI)
14
+ ```
15
+
16
+ Or work in it live, the way you would in Colab:
17
+
18
+ ```bash
19
+ testbench session start # or: --backend modal --gpu T4, or --backend colab
20
+ testbench exec --cells tiny_model.ipynb:1-3 --params '{"EPOCHS": 10}'
21
+ testbench exec "model.n_iter_, model.score(X_test, y_test)"
22
+ testbench session stop
23
+ ```
24
+
25
+ Try breaking it (`params: {HIDDEN: 0}` in a plan, or `LEARNING_RATE = 1.0` in `cells/02_settings.py` and
26
+ `testbench nb build cells tiny_model.ipynb`) and read what the report says.
@@ -0,0 +1,7 @@
1
+ # web-ui
2
+
3
+ A one-page counter checked the way a person would: `testbench run ui-check` serves `site/`, opens it in headless
4
+ Chromium, clicks "Add one" twice, takes screenshots, and checks the page says `Count: 2` with no console errors.
5
+ Needs `pip install "agent-testbench[web]"` and `playwright install chromium`.
6
+
7
+ Break `site/app.js` (call a function that does not exist) and the report names the page error.