agent-testbench 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. agent_testbench/__init__.py +3 -0
  2. agent_testbench/__main__.py +3 -0
  3. agent_testbench/backends/__init__.py +59 -0
  4. agent_testbench/backends/colab_backend.py +157 -0
  5. agent_testbench/backends/local.py +25 -0
  6. agent_testbench/backends/modal_backend.py +212 -0
  7. agent_testbench/budget.py +113 -0
  8. agent_testbench/cli.py +514 -0
  9. agent_testbench/colab_shim/README.md +5 -0
  10. agent_testbench/colab_shim/google/colab/__init__.py +3 -0
  11. agent_testbench/colab_shim/google/colab/drive.py +11 -0
  12. agent_testbench/colab_shim/google/colab/files.py +45 -0
  13. agent_testbench/colab_shim/google/colab/output.py +25 -0
  14. agent_testbench/colab_shim/google/colab/patches.py +10 -0
  15. agent_testbench/colab_shim/google/colab/userdata.py +23 -0
  16. agent_testbench/config.py +188 -0
  17. agent_testbench/driver.py +72 -0
  18. agent_testbench/executor.py +255 -0
  19. agent_testbench/expect.py +156 -0
  20. agent_testbench/kernelkit/__init__.py +3 -0
  21. agent_testbench/kernelkit/client.py +138 -0
  22. agent_testbench/kernelkit/launcher.py +61 -0
  23. agent_testbench/lint.py +82 -0
  24. agent_testbench/mcp_server.py +169 -0
  25. agent_testbench/mocks.py +74 -0
  26. agent_testbench/notebook.py +274 -0
  27. agent_testbench/report.py +129 -0
  28. agent_testbench/runs.py +236 -0
  29. agent_testbench/session_backends.py +471 -0
  30. agent_testbench/sessions.py +298 -0
  31. agent_testbench/web.py +129 -0
  32. agent_testbench-0.1.0.dist-info/METADATA +301 -0
  33. agent_testbench-0.1.0.dist-info/RECORD +36 -0
  34. agent_testbench-0.1.0.dist-info/WHEEL +4 -0
  35. agent_testbench-0.1.0.dist-info/entry_points.txt +3 -0
  36. agent_testbench-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,3 @@
1
+ """agent-testbench: a test bench your coding agent operates - live GPU sessions and verified runs on Modal,
2
+ Colab or this machine, inside a budget."""
3
+ __version__ = "0.1.0"
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ main()
@@ -0,0 +1,59 @@
1
+ """Where a plan runs. Each backend takes a launched run and returns (report, cost in US dollars)."""
2
+ from __future__ import annotations
3
+
4
+ import fnmatch
5
+ import shutil
6
+ from pathlib import Path
7
+
8
+
9
+ def get_backend(name: str):
10
+ if name == "local":
11
+ from .local import LocalBackend
12
+ return LocalBackend
13
+ if name == "modal":
14
+ from .modal_backend import ModalBackend
15
+ return ModalBackend
16
+ if name == "colab":
17
+ from .colab_backend import ColabBackend
18
+ return ColabBackend
19
+ raise ValueError(f"unknown backend {name!r}")
20
+
21
+
22
+ def cancel_remote(state: dict) -> str:
23
+ """Cancel remote work for a run whose driver is gone, from the handles saved in state.json."""
24
+ if state.get("modal_call_id"):
25
+ from .modal_backend import cancel_call
26
+ return cancel_call(state["modal_call_id"])
27
+ if state.get("colab_session"):
28
+ from .colab_backend import stop_session
29
+ return stop_session(state["colab_session"], state.get("colab_cli_args", []))
30
+ return ""
31
+
32
+
33
+ def excluded(rel: str, patterns: list[str]) -> bool:
34
+ parts = Path(rel).parts
35
+ return any(fnmatch.fnmatch(rel, p) or any(fnmatch.fnmatch(part, p) for part in parts) for p in patterns)
36
+
37
+
38
+ def copy_project(root: Path, dest: Path, exclude: list[str]) -> Path:
39
+ """A scratch copy of the project, without the excluded files (and never .testbench itself)."""
40
+ root, dest = Path(root), Path(dest)
41
+ patterns = list(exclude) + [".testbench"]
42
+
43
+ def ignore(directory, names):
44
+ rel_dir = Path(directory).relative_to(root)
45
+ return [n for n in names if excluded(str(rel_dir / n) if str(rel_dir) != "." else n, patterns)]
46
+
47
+ shutil.copytree(root, dest, ignore=ignore, dirs_exist_ok=True, symlinks=True)
48
+ return dest
49
+
50
+
51
+ def project_files(root: Path, exclude: list[str]) -> list[Path]:
52
+ root = Path(root)
53
+ patterns = list(exclude) + [".testbench"]
54
+ out = []
55
+ for p in sorted(root.rglob("*")):
56
+ rel = str(p.relative_to(root))
57
+ if p.is_file() and not excluded(rel, patterns):
58
+ out.append(p)
59
+ return out
@@ -0,0 +1,157 @@
1
+ """Run on a Colab VM through Google's Colab CLI (`uv tool install google-colab-cli`).
2
+
3
+ Per run: allocate a session (CPU or the plan's GPU), install the pinned basics, upload the project as one
4
+ bundle (in chunks - uploads over ~100 MB fail), execute the plan with a timeout equal to its budget, bring
5
+ the run folder back, and always stop the session, so compute units stop with it. The project lands in
6
+ /content, where Colab notebooks expect their files.
7
+
8
+ Colab notes: VMs can be reclaimed after a couple of hours even while busy, so keep Colab plans short and
9
+ use Modal for long jobs; set `colab: {auth: adc}` in testbench.yaml if you log in with Application Default
10
+ Credentials.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import io
15
+ import json
16
+ import shutil
17
+ import subprocess
18
+ import tarfile
19
+ import tempfile
20
+ import time
21
+ from pathlib import Path
22
+
23
+ from .. import budget
24
+ from . import project_files
25
+
26
+ CHUNK = 45 * 1024 * 1024
27
+ PKG = Path(__file__).resolve().parents[1] # the agent_testbench package, shipped with the project
28
+
29
+ BOOTSTRAP = r'''
30
+ import glob, json, os, subprocess, sys, tarfile
31
+ parts = sorted(glob.glob("/content/.testbench_bundle.part*"))
32
+ with open("/content/.testbench_bundle.tar.gz", "wb") as out:
33
+ for p in parts:
34
+ with open(p, "rb") as f:
35
+ out.write(f.read())
36
+ os.remove(p)
37
+ safe = {"filter": "data"} if hasattr(tarfile, "data_filter") else {}
38
+ with tarfile.open("/content/.testbench_bundle.tar.gz") as t:
39
+ for m in t.getmembers():
40
+ if m.name.startswith("project/"):
41
+ m.name = m.name[len("project/"):]
42
+ if m.name:
43
+ t.extract(m, "/content", **safe)
44
+ elif m.name.startswith("pkg/"):
45
+ m.name = m.name[len("pkg/"):]
46
+ t.extract(m, "/content/.testbench_pkg", **safe)
47
+ sys.path.insert(0, "/content/.testbench_pkg")
48
+ os.environ["TESTBENCH_CACHE"] = "/content/.testbench_cache"
49
+ os.makedirs("/content/.testbench_cache", exist_ok=True)
50
+ from agent_testbench.executor import execute_plan
51
+ run_dir = "/content/.testbench_run/" + RUN_ID
52
+ report = execute_plan(PLAN, project_dir="/content", run_dir=run_dir, workdir="/content", meta=META)
53
+ print("TESTBENCH_DONE " + json.dumps({"status": report["status"], "minutes": report.get("minutes")}), flush=True)
54
+ '''
55
+
56
+ PACK = r'''
57
+ import os, tarfile
58
+ d = "/content/.testbench_run/" + RUN_ID
59
+ with tarfile.open("/content/.testbench_result.tar.gz", "w:gz") as t:
60
+ if os.path.isdir(d):
61
+ t.add(d, arcname=".")
62
+ print("TESTBENCH_PACKED", os.path.getsize("/content/.testbench_result.tar.gz"), flush=True)
63
+ '''
64
+
65
+
66
+ def cli_args(cfg: dict) -> list[str]:
67
+ c = cfg.get("colab") or {}
68
+ return [c.get("cli", "colab")] + (["--auth", c["auth"]] if c.get("auth") else [])
69
+
70
+
71
+ def stop_session(session: str, args: list[str]) -> str:
72
+ try:
73
+ subprocess.run([*(args or ["colab"]), "stop", "-s", session], capture_output=True, text=True, timeout=120)
74
+ return f"Colab session {session} stopped"
75
+ except Exception as exc:
76
+ return f"could not stop Colab session {session}: {exc} - stop it with `colab stop -s {session}`"
77
+
78
+
79
+ class ColabBackend:
80
+ def __init__(self, root: Path, cfg: dict, run_dir: Path, plan: dict, meta: dict):
81
+ self.root, self.cfg, self.run_dir, self.plan, self.meta = Path(root), cfg, Path(run_dir), plan, meta
82
+ self.args = cli_args(cfg)
83
+ self.session = ("testbench-" + run_dir.name)[:60]
84
+ self.started = None
85
+
86
+ def _colab(self, *argv, timeout: float = 900, check: bool = True, input: str | None = None):
87
+ cmd = [*self.args, *argv]
88
+ proc = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout, input=input)
89
+ with (self.run_dir / "colab.log").open("a") as log:
90
+ log.write(f"$ {' '.join(cmd[:4])} ...\n{proc.stdout[-4000:]}{proc.stderr[-4000:]}\n")
91
+ if check and proc.returncode != 0:
92
+ raise RuntimeError(f"`colab {argv[0]}` failed (exit {proc.returncode}): "
93
+ f"{(proc.stderr or proc.stdout).strip().splitlines()[-1:] or ''}. See colab.log in the run folder.")
94
+ return proc
95
+
96
+ def _bundle(self) -> Path:
97
+ path = Path(tempfile.mkdtemp()) / "bundle.tar.gz"
98
+ with tarfile.open(path, "w:gz") as t:
99
+ for f in project_files(self.root, self.cfg["exclude"]):
100
+ t.add(f, arcname="project/" + str(f.relative_to(self.root)))
101
+ t.add(PKG, arcname="pkg/agent_testbench", filter=lambda m: None if "__pycache__" in m.name else m)
102
+ return path
103
+
104
+ def run(self):
105
+ from .. import runs, report as R
106
+ if not shutil.which(self.args[0]):
107
+ raise RuntimeError("The Colab CLI is not installed: `uv tool install google-colab-cli`, then log in once "
108
+ "with `colab new` (or set colab.auth: adc in testbench.yaml).")
109
+ gpu = (self.plan.get("gpu") or "").upper()
110
+ runs.write_state(self.run_dir, colab_session=self.session, colab_cli_args=self.args)
111
+ print(f"[testbench] allocating Colab session {self.session} ({gpu or 'CPU'})", flush=True)
112
+ self._colab("new", "-s", self.session, *(["--gpu", gpu] if gpu else []), timeout=900)
113
+ self.started = time.time()
114
+ try:
115
+ pkgs = ["nbformat>=5.9", "nbclient>=0.10", "ipykernel>=6.29", "pyyaml>=6", *self.cfg["colab"].get("pip", [])]
116
+ self._colab("install", "-s", self.session, *pkgs, timeout=1200)
117
+ if self.cfg["colab"].get("requirements"):
118
+ self._colab("install", "-s", self.session, "-r", str(self.root / self.cfg["colab"]["requirements"]), timeout=1800)
119
+ bundle = self._bundle()
120
+ data = bundle.read_bytes()
121
+ for i in range(0, max(1, len(data)), CHUNK):
122
+ part = bundle.with_name(f"part{i // CHUNK:03d}")
123
+ part.write_bytes(data[i:i + CHUNK])
124
+ self._colab("upload", "-s", self.session, str(part), f"/content/.testbench_bundle.part{i // CHUNK:03d}", timeout=900)
125
+ minutes = sum(s["max_minutes"] for s in self.plan["steps"])
126
+ script = (f"RUN_ID = {self.run_dir.name!r}\nPLAN = json.loads({json.dumps(self.plan)!r})\n"
127
+ f"META = json.loads({json.dumps(self.meta)!r})\n")
128
+ boot = Path(tempfile.mkdtemp()) / "testbench_boot.py"
129
+ boot.write_text("import json\n" + script + BOOTSTRAP)
130
+ print(f"[testbench] running the plan on Colab (up to {minutes:.0f} min)", flush=True)
131
+ proc = self._colab("exec", "-s", self.session, "-f", str(boot), "--timeout", str(int(minutes * 60 + 120)),
132
+ timeout=minutes * 60 + 600, check=False)
133
+ (self.run_dir / "colab_exec.log").write_text(proc.stdout + proc.stderr)
134
+ pack = Path(tempfile.mkdtemp()) / "testbench_pack.py"
135
+ pack.write_text(f"RUN_ID = {self.run_dir.name!r}\n" + PACK)
136
+ self._colab("exec", "-s", self.session, "-f", str(pack), "--timeout", "300", timeout=600)
137
+ local = Path(tempfile.mkdtemp()) / "result.tar.gz"
138
+ self._colab("download", "-s", self.session, "/content/.testbench_result.tar.gz", str(local), timeout=1800)
139
+ with tarfile.open(local) as t:
140
+ t.extractall(self.run_dir, **({"filter": "data"} if hasattr(tarfile, "data_filter") else {}))
141
+ finally:
142
+ print(f"[testbench] {stop_session(self.session, self.args)}", flush=True)
143
+ rep = R.load(self.run_dir)
144
+ if not rep:
145
+ log = (self.run_dir / "colab_exec.log")
146
+ tail = "\n".join(log.read_text(errors="replace").strip().splitlines()[-20:]) if log.exists() else ""
147
+ rep = {**self.meta, "steps": [], "started": self.started, "status": "error",
148
+ "stopped_because": "the plan never started on the Colab VM",
149
+ "launch_error": "The VM could not start the plan. Its last output:\n\n```\n" + tail + "\n```"}
150
+ return rep, self.cost(rep)
151
+
152
+ def cancel(self) -> str:
153
+ return stop_session(self.session, self.args)
154
+
155
+ def cost(self, rep: dict) -> float:
156
+ minutes = (time.time() - self.started) / 60 if self.started else 0.0
157
+ return budget.rate_per_minute(self.cfg, self.plan) * minutes
@@ -0,0 +1,25 @@
1
+ """Run on this machine: free, immediate, and the first rung of every ladder."""
2
+ from __future__ import annotations
3
+
4
+ from pathlib import Path
5
+
6
+ from .. import executor
7
+ from . import copy_project
8
+
9
+
10
+ class LocalBackend:
11
+ def __init__(self, root: Path, cfg: dict, run_dir: Path, plan: dict, meta: dict):
12
+ self.root, self.cfg, self.run_dir, self.plan, self.meta = Path(root), cfg, Path(run_dir), plan, meta
13
+
14
+ def run(self):
15
+ workdir = self.root
16
+ if self.plan.get("workdir") == "copy":
17
+ workdir = copy_project(self.root, self.run_dir / "work", self.cfg["exclude"])
18
+ rep = executor.execute_plan(self.plan, project_dir=self.root, run_dir=self.run_dir, workdir=workdir, meta=self.meta)
19
+ return rep, 0.0
20
+
21
+ def cancel(self) -> str:
22
+ return ""
23
+
24
+ def cost(self, rep: dict) -> float:
25
+ return 0.0
@@ -0,0 +1,212 @@
1
+ """Run on Modal: one deployed app per project, one function per plan, results on a Modal volume.
2
+
3
+ The app is generated from testbench.yaml into .testbench/modal_app.py (readable, not hand-edited), deployed, and
4
+ the plan's function is spawned, so the run keeps going if this machine sleeps or loses its network. Lessons
5
+ built in:
6
+ - the function's timeout is the plan's budget, so the worst case holds on Modal's side;
7
+ - CPU and memory are always explicit (a GPU function without cpu= gets one core, and the GPU waits on it);
8
+ - no retries and a 2-second scaledown, so a finished or failed run stops billing at once;
9
+ - local files are added as the image's last steps, so editing code does not rebuild the image.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ import subprocess
15
+ import sys
16
+ import textwrap
17
+ import threading
18
+ import time
19
+ from pathlib import Path
20
+
21
+ from .. import budget
22
+
23
+ APP_TEMPLATE = '''\
24
+ # Generated by agent-testbench from testbench.yaml - edit testbench.yaml, not this file. Regenerated on every Modal run.
25
+ import json
26
+ import threading
27
+ import time
28
+ from pathlib import Path
29
+
30
+ import modal
31
+
32
+ app = modal.App({app_name!r})
33
+ volume = modal.Volume.from_name({volume_name!r}, create_if_missing=True)
34
+ image = (
35
+ modal.Image.debian_slim(python_version={python!r})
36
+ {image_steps}
37
+ .add_local_python_source("agent_testbench")
38
+ .add_local_dir({root!r}, "/project", ignore={ignore!r})
39
+ )
40
+
41
+
42
+ def _run(run_id: str, plan: dict, meta: dict) -> dict:
43
+ import os
44
+ import shutil
45
+ from agent_testbench.executor import execute_plan
46
+
47
+ work = Path("/work/project")
48
+ shutil.copytree("/project", work, dirs_exist_ok=True, symlinks=True)
49
+ if not Path("/content").exists(): # Colab notebooks expect to live in /content
50
+ Path("/content").symlink_to(work)
51
+ os.environ["TESTBENCH_CACHE"] = "/testbench/cache"
52
+ Path("/testbench/cache").mkdir(parents=True, exist_ok=True)
53
+ run_dir = Path("/testbench/runs") / run_id
54
+ stop = threading.Event()
55
+
56
+ def keep_committing(): # live logs reach the volume about once a minute
57
+ while not stop.wait(60):
58
+ try:
59
+ volume.commit()
60
+ except Exception:
61
+ pass
62
+
63
+ threading.Thread(target=keep_committing, daemon=True).start()
64
+ try:
65
+ report = execute_plan(plan, project_dir=work, run_dir=run_dir, workdir=work, meta=meta, on_progress=volume.commit)
66
+ finally:
67
+ stop.set()
68
+ volume.commit()
69
+ return {{"status": report["status"], "minutes": report.get("minutes")}}
70
+ {functions}
71
+ '''
72
+
73
+ FUNCTION_TEMPLATE = '''
74
+
75
+ @app.function(image=image, gpu={gpu!r}, cpu={cpu!r}, memory={memory!r}, timeout={timeout!r}, retries=0,
76
+ scaledown_window=2, volumes={{"/testbench": volume}}, secrets=[{secrets}])
77
+ def {fn}(run_id: str, plan: dict, meta: dict) -> dict:
78
+ return _run(run_id, plan, meta)
79
+ '''
80
+
81
+
82
+ def app_name(cfg: dict) -> str:
83
+ return f"testbench-{cfg['project']}"
84
+
85
+
86
+ def fn_name(plan_name: str) -> str:
87
+ return "plan_" + "".join(c if c.isalnum() else "_" for c in plan_name)
88
+
89
+
90
+ def generate(cfg: dict) -> str:
91
+ m = cfg["modal"]
92
+ steps = []
93
+ if m.get("apt"):
94
+ steps.append(f" .apt_install({', '.join(repr(x) for x in m['apt'])})")
95
+ base = ["nbformat>=5.9", "nbclient>=0.10", "ipykernel>=6.29", "pyyaml>=6"]
96
+ uses_web = any(s["kind"] == "web" for p in cfg["plans"].values() if p["backend"] == "modal" for s in p["steps"])
97
+ if uses_web:
98
+ base.append("playwright>=1.45")
99
+ steps.append(f" .pip_install({', '.join(repr(x) for x in base + list(m.get('pip') or []))})")
100
+ if m.get("requirements"):
101
+ req = Path(cfg["root"]) / m["requirements"]
102
+ steps.append(f" .pip_install_from_requirements({str(req)!r})")
103
+ if uses_web:
104
+ steps.append(' .run_commands("playwright install --with-deps chromium")')
105
+ for cmd in m.get("run_commands") or []:
106
+ steps.append(f" .run_commands({cmd!r})")
107
+ ignore = []
108
+ for p in list(cfg["exclude"]) + [".testbench"]:
109
+ ignore += [p] if "/" in p else [p, f"**/{p}"]
110
+ functions = []
111
+ for name, plan in cfg["plans"].items():
112
+ if plan["backend"] != "modal":
113
+ continue
114
+ minutes = sum(s["max_minutes"] for s in plan["steps"])
115
+ secrets = sorted({s for st in plan["steps"] for s in st["secrets"]})
116
+ functions.append(FUNCTION_TEMPLATE.format(
117
+ gpu=plan.get("gpu"), cpu=float(plan.get("cpu", m["cpu"])), memory=int(float(plan.get("memory_gb", m["memory_gb"])) * 1024),
118
+ timeout=int((minutes + 3) * 60), fn=fn_name(name),
119
+ secrets=", ".join(f"modal.Secret.from_name({s!r})" for s in secrets)))
120
+ return APP_TEMPLATE.format(app_name=app_name(cfg), volume_name=app_name(cfg), python=str(m["python"]),
121
+ image_steps="\n".join(steps), root=str(cfg["root"]), ignore=ignore,
122
+ functions="".join(functions))
123
+
124
+
125
+ def cancel_call(call_id: str) -> str:
126
+ import modal
127
+ try:
128
+ modal.FunctionCall.from_id(call_id).cancel(terminate_containers=True)
129
+ return f"Modal call {call_id} cancelled"
130
+ except Exception as exc:
131
+ return f"could not cancel Modal call {call_id}: {exc}"
132
+
133
+
134
+ class ModalBackend:
135
+ def __init__(self, root: Path, cfg: dict, run_dir: Path, plan: dict, meta: dict):
136
+ self.root, self.cfg, self.run_dir, self.plan, self.meta = Path(root), cfg, Path(run_dir), plan, meta
137
+ self.call = None
138
+ self.remote_started = None
139
+ self._seen: dict[str, tuple] = {}
140
+
141
+ def _log(self, text: str) -> None:
142
+ print(f"[testbench] {text}", flush=True)
143
+
144
+ def deploy(self) -> None:
145
+ app_file = self.root / ".testbench" / "modal_app.py"
146
+ app_file.parent.mkdir(parents=True, exist_ok=True)
147
+ app_file.write_text(generate(self.cfg))
148
+ self._log(f"deploying {app_name(self.cfg)} from {app_file}")
149
+ proc = subprocess.run([sys.executable, "-m", "modal", "deploy", str(app_file)], cwd=self.root,
150
+ capture_output=True, text=True, timeout=1800)
151
+ (self.run_dir / "modal_deploy.log").write_text(proc.stdout + proc.stderr)
152
+ if proc.returncode != 0:
153
+ tail = "\n".join((proc.stdout + proc.stderr).strip().splitlines()[-25:])
154
+ raise RuntimeError(f"`modal deploy` failed (is Modal set up? run `modal setup`):\n{tail}")
155
+
156
+ def run(self):
157
+ import modal
158
+ from .. import runs
159
+ self.deploy()
160
+ fn = modal.Function.from_name(app_name(self.cfg), fn_name(self.meta["plan"]))
161
+ self.call = fn.spawn(self.run_dir.name, self.plan, self.meta)
162
+ self.remote_started = time.time()
163
+ runs.write_state(self.run_dir, modal_call_id=self.call.object_id, modal_app=app_name(self.cfg))
164
+ self._log(f"spawned Modal call {self.call.object_id}")
165
+ volume = modal.Volume.from_name(app_name(self.cfg))
166
+ result = None
167
+ while True:
168
+ try:
169
+ result = self.call.get(timeout=20)
170
+ break
171
+ except (TimeoutError, modal.exception.TimeoutError):
172
+ self._mirror(volume, light=True)
173
+ except modal.exception.FunctionTimeoutError:
174
+ result = {"status": "timeout"}
175
+ break
176
+ self._mirror(volume, light=False)
177
+ from .. import report as R
178
+ rep = R.load(self.run_dir) or {**self.meta, "steps": [], "started": self.remote_started, "status": "error"}
179
+ if result and result.get("status") == "timeout":
180
+ rep.update(status="failed", stopped_because="the plan ran past its total max_minutes on Modal")
181
+ return rep, self.cost(rep)
182
+
183
+ def _mirror(self, volume, light: bool) -> None:
184
+ """Copy the run's files from the volume: progress and logs while it runs, everything at the end."""
185
+ prefix = f"runs/{self.run_dir.name}"
186
+ try:
187
+ entries = volume.listdir(prefix, recursive=True)
188
+ except Exception:
189
+ return
190
+ for e in entries:
191
+ if getattr(e.type, "name", str(e.type)) != "FILE":
192
+ continue
193
+ rel = Path(e.path).relative_to(prefix)
194
+ if light and not (rel.name in ("progress.jsonl", "report.json", "log.txt", "result.json")):
195
+ continue
196
+ stamp = (getattr(e, "size", None), getattr(e, "mtime", None))
197
+ if self._seen.get(e.path) == stamp and stamp != (None, None):
198
+ continue # unchanged since the last copy
199
+ self._seen[e.path] = stamp
200
+ target = self.run_dir / rel
201
+ target.parent.mkdir(parents=True, exist_ok=True)
202
+ try:
203
+ target.write_bytes(b"".join(volume.read_file(e.path)))
204
+ except Exception:
205
+ pass
206
+
207
+ def cancel(self) -> str:
208
+ return cancel_call(self.call.object_id) if self.call else ""
209
+
210
+ def cost(self, rep: dict) -> float:
211
+ minutes = (time.time() - self.remote_started) / 60 if self.remote_started else 0.0
212
+ return budget.rate_per_minute(self.cfg, self.plan) * (minutes + 0.5)
@@ -0,0 +1,113 @@
1
+ """Money: what a plan can cost at worst, what runs have cost, and whether a launch is allowed.
2
+
3
+ The worst case of a plan is its hardware's rate times the sum of its steps' max_minutes. Remote steps are
4
+ also stopped on the remote side at that limit (a Modal function timeout, a Colab exec timeout), so the
5
+ estimate is a bound, not a hope - it holds even if this machine goes offline mid-run.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import datetime as dt
10
+ import json
11
+ from pathlib import Path
12
+
13
+ from . import config as C
14
+
15
+ # Modal, US dollars per second (modal.com/pricing, checked 2026-10). Override any of them under `pricing:`.
16
+ MODAL_GPU_PER_S = {"B300": 0.001972, "B200": 0.001736, "H200": 0.001261, "H100": 0.001097, "RTX-PRO-6000": 0.000842,
17
+ "A100-80GB": 0.000694, "A100-40GB": 0.000583, "A100": 0.000583, "L40S": 0.000542, "A10": 0.000306,
18
+ "A10G": 0.000306, "L4": 0.000222, "T4": 0.000164}
19
+ MODAL_CPU_CORE_PER_S = 0.0000131
20
+ MODAL_MEM_GIB_PER_S = 0.00000222
21
+ # Colab bills compute units. These per-hour rates are approximate and vary by account and region; `colab usage`
22
+ # shows your actual rate - put it under `pricing: colab_units_per_hour:`. Pay-as-you-go is ~$10 per 100 units.
23
+ COLAB_UNITS_PER_HOUR = {"CPU": 0.1, "T4": 1.8, "L4": 2.0, "G4": 4.0, "A100": 5.4, "H100": 9.0}
24
+ COLAB_USD_PER_UNIT = 0.10
25
+
26
+
27
+ def rate_per_minute(cfg: dict, plan: dict) -> float:
28
+ """US dollars per minute for the plan's hardware; 0 for local plans."""
29
+ prices = cfg.get("pricing") or {}
30
+ backend = plan["backend"]
31
+ if backend == "local":
32
+ return 0.0
33
+ gpu = (plan.get("gpu") or "").split(":")[0].upper()
34
+ count = int(plan["gpu"].split(":")[1]) if plan.get("gpu") and ":" in plan["gpu"] else 1
35
+ if backend == "modal":
36
+ gpu_rates = {**MODAL_GPU_PER_S, **(prices.get("modal_gpu_per_s") or {})}
37
+ cpu = float(plan.get("cpu") or cfg["modal"]["cpu"])
38
+ mem = float(plan.get("memory_gb") or cfg["modal"]["memory_gb"])
39
+ per_s = (gpu_rates.get(gpu, 0.0) * count if gpu else 0.0) \
40
+ + cpu * float(prices.get("modal_cpu_core_per_s", MODAL_CPU_CORE_PER_S)) \
41
+ + mem * float(prices.get("modal_mem_gib_per_s", MODAL_MEM_GIB_PER_S))
42
+ return per_s * 60
43
+ units = {**COLAB_UNITS_PER_HOUR, **(prices.get("colab_units_per_hour") or {})}
44
+ return units.get(gpu or "CPU", 0.0) * float(prices.get("colab_usd_per_unit", COLAB_USD_PER_UNIT)) / 60
45
+
46
+
47
+ def estimate(cfg: dict, hardware: dict, minutes: float, label: str) -> dict:
48
+ """Worst case for `minutes` on `hardware` ({backend, gpu, cpu?, memory_gb?}); a remote machine also pays for
49
+ starting up and installing, so a few minutes are added."""
50
+ rate = rate_per_minute(cfg, hardware)
51
+ overhead = 0 if hardware["backend"] == "local" else 5
52
+ return {"what": label, "backend": hardware["backend"], "gpu": hardware.get("gpu"), "usd_per_minute": rate,
53
+ "max_minutes": float(minutes), "worst_case_usd": round(rate * (float(minutes) + overhead), 4)}
54
+
55
+
56
+ def worst_case(cfg: dict, plan_name: str) -> dict:
57
+ plan = cfg["plans"][plan_name]
58
+ return {**estimate(cfg, plan, C.plan_minutes(plan), f"plan {plan_name}"), "plan": plan_name}
59
+
60
+
61
+ def ledger_path(root: Path) -> Path:
62
+ return Path(root) / ".testbench" / "ledger.jsonl"
63
+
64
+
65
+ def record(root: Path, entry: dict) -> None:
66
+ p = ledger_path(root)
67
+ p.parent.mkdir(parents=True, exist_ok=True)
68
+ with p.open("a") as f:
69
+ f.write(json.dumps({"time": dt.datetime.now().isoformat(timespec="seconds"), **entry}) + "\n")
70
+
71
+
72
+ def entries(root: Path) -> list[dict]:
73
+ p = ledger_path(root)
74
+ if not p.exists():
75
+ return []
76
+ return [json.loads(line) for line in p.read_text().splitlines() if line.strip()]
77
+
78
+
79
+ def spent_today(root: Path) -> float:
80
+ """Dollars committed today: finished runs at their measured cost, unfinished ones at their worst case."""
81
+ today = dt.date.today().isoformat()
82
+ runs: dict[str, float] = {}
83
+ for e in entries(root):
84
+ if not e["time"].startswith(today):
85
+ continue
86
+ if e["event"] == "launch":
87
+ runs[e["run"]] = e["worst_case_usd"]
88
+ elif e["event"] == "finish":
89
+ runs[e["run"]] = e["usd"]
90
+ return round(sum(runs.values()), 4)
91
+
92
+
93
+ def check(cfg: dict, est: dict, approve: bool, how_to_lower: str) -> tuple[bool, str, dict]:
94
+ """(allowed, message, estimate). Over max_run_usd or the day's budget: refused. Over ask_above_usd: needs
95
+ approve=True, which an agent may only pass after a person said yes."""
96
+ b = cfg["budget"]
97
+ cost, what = est["worst_case_usd"], est["what"]
98
+ if cost > b["max_run_usd"]:
99
+ return False, (f"Refused: {what} could cost up to ${cost:.2f} ({est['max_minutes']:.0f} min at "
100
+ f"${est['usd_per_minute']:.4f}/min), over budget.max_run_usd ${b['max_run_usd']:.2f}. "
101
+ f"{how_to_lower}, use a cheaper GPU, or have a person raise the budget."), est
102
+ today = spent_today(Path(cfg["root"]))
103
+ if today + cost > b["max_day_usd"]:
104
+ return False, (f"Refused: ${today:.2f} already committed today; {what} could add ${cost:.2f}, over "
105
+ f"budget.max_day_usd ${b['max_day_usd']:.2f}."), est
106
+ if cost > b["ask_above_usd"] and not approve:
107
+ return False, (f"Needs approval: {what} could cost up to ${cost:.2f} (above budget.ask_above_usd "
108
+ f"${b['ask_above_usd']:.2f}). Ask a person; if they agree, rerun with --approve."), est
109
+ return True, f"Within budget: up to ${cost:.2f} (today so far ${today:.2f} of ${b['max_day_usd']:.2f}).", est
110
+
111
+
112
+ def check_launch(cfg: dict, plan_name: str, approve: bool) -> tuple[bool, str, dict]:
113
+ return check(cfg, worst_case(cfg, plan_name), approve, "Lower the steps' max_minutes")