agent-testbench 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_testbench/__init__.py +3 -0
- agent_testbench/__main__.py +3 -0
- agent_testbench/backends/__init__.py +59 -0
- agent_testbench/backends/colab_backend.py +157 -0
- agent_testbench/backends/local.py +25 -0
- agent_testbench/backends/modal_backend.py +212 -0
- agent_testbench/budget.py +113 -0
- agent_testbench/cli.py +514 -0
- agent_testbench/colab_shim/README.md +5 -0
- agent_testbench/colab_shim/google/colab/__init__.py +3 -0
- agent_testbench/colab_shim/google/colab/drive.py +11 -0
- agent_testbench/colab_shim/google/colab/files.py +45 -0
- agent_testbench/colab_shim/google/colab/output.py +25 -0
- agent_testbench/colab_shim/google/colab/patches.py +10 -0
- agent_testbench/colab_shim/google/colab/userdata.py +23 -0
- agent_testbench/config.py +188 -0
- agent_testbench/driver.py +72 -0
- agent_testbench/executor.py +255 -0
- agent_testbench/expect.py +156 -0
- agent_testbench/kernelkit/__init__.py +3 -0
- agent_testbench/kernelkit/client.py +138 -0
- agent_testbench/kernelkit/launcher.py +61 -0
- agent_testbench/lint.py +82 -0
- agent_testbench/mcp_server.py +169 -0
- agent_testbench/mocks.py +74 -0
- agent_testbench/notebook.py +274 -0
- agent_testbench/report.py +129 -0
- agent_testbench/runs.py +236 -0
- agent_testbench/session_backends.py +471 -0
- agent_testbench/sessions.py +298 -0
- agent_testbench/web.py +129 -0
- agent_testbench-0.1.0.dist-info/METADATA +301 -0
- agent_testbench-0.1.0.dist-info/RECORD +36 -0
- agent_testbench-0.1.0.dist-info/WHEEL +4 -0
- agent_testbench-0.1.0.dist-info/entry_points.txt +3 -0
- agent_testbench-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""Where a plan runs. Each backend takes a launched run and returns (report, cost in US dollars)."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import fnmatch
|
|
5
|
+
import shutil
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def get_backend(name: str):
|
|
10
|
+
if name == "local":
|
|
11
|
+
from .local import LocalBackend
|
|
12
|
+
return LocalBackend
|
|
13
|
+
if name == "modal":
|
|
14
|
+
from .modal_backend import ModalBackend
|
|
15
|
+
return ModalBackend
|
|
16
|
+
if name == "colab":
|
|
17
|
+
from .colab_backend import ColabBackend
|
|
18
|
+
return ColabBackend
|
|
19
|
+
raise ValueError(f"unknown backend {name!r}")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def cancel_remote(state: dict) -> str:
|
|
23
|
+
"""Cancel remote work for a run whose driver is gone, from the handles saved in state.json."""
|
|
24
|
+
if state.get("modal_call_id"):
|
|
25
|
+
from .modal_backend import cancel_call
|
|
26
|
+
return cancel_call(state["modal_call_id"])
|
|
27
|
+
if state.get("colab_session"):
|
|
28
|
+
from .colab_backend import stop_session
|
|
29
|
+
return stop_session(state["colab_session"], state.get("colab_cli_args", []))
|
|
30
|
+
return ""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def excluded(rel: str, patterns: list[str]) -> bool:
|
|
34
|
+
parts = Path(rel).parts
|
|
35
|
+
return any(fnmatch.fnmatch(rel, p) or any(fnmatch.fnmatch(part, p) for part in parts) for p in patterns)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def copy_project(root: Path, dest: Path, exclude: list[str]) -> Path:
|
|
39
|
+
"""A scratch copy of the project, without the excluded files (and never .testbench itself)."""
|
|
40
|
+
root, dest = Path(root), Path(dest)
|
|
41
|
+
patterns = list(exclude) + [".testbench"]
|
|
42
|
+
|
|
43
|
+
def ignore(directory, names):
|
|
44
|
+
rel_dir = Path(directory).relative_to(root)
|
|
45
|
+
return [n for n in names if excluded(str(rel_dir / n) if str(rel_dir) != "." else n, patterns)]
|
|
46
|
+
|
|
47
|
+
shutil.copytree(root, dest, ignore=ignore, dirs_exist_ok=True, symlinks=True)
|
|
48
|
+
return dest
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def project_files(root: Path, exclude: list[str]) -> list[Path]:
|
|
52
|
+
root = Path(root)
|
|
53
|
+
patterns = list(exclude) + [".testbench"]
|
|
54
|
+
out = []
|
|
55
|
+
for p in sorted(root.rglob("*")):
|
|
56
|
+
rel = str(p.relative_to(root))
|
|
57
|
+
if p.is_file() and not excluded(rel, patterns):
|
|
58
|
+
out.append(p)
|
|
59
|
+
return out
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
"""Run on a Colab VM through Google's Colab CLI (`uv tool install google-colab-cli`).
|
|
2
|
+
|
|
3
|
+
Per run: allocate a session (CPU or the plan's GPU), install the pinned basics, upload the project as one
|
|
4
|
+
bundle (in chunks - uploads over ~100 MB fail), execute the plan with a timeout equal to its budget, bring
|
|
5
|
+
the run folder back, and always stop the session, so compute units stop with it. The project lands in
|
|
6
|
+
/content, where Colab notebooks expect their files.
|
|
7
|
+
|
|
8
|
+
Colab notes: VMs can be reclaimed after a couple of hours even while busy, so keep Colab plans short and
|
|
9
|
+
use Modal for long jobs; set `colab: {auth: adc}` in testbench.yaml if you log in with Application Default
|
|
10
|
+
Credentials.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import io
|
|
15
|
+
import json
|
|
16
|
+
import shutil
|
|
17
|
+
import subprocess
|
|
18
|
+
import tarfile
|
|
19
|
+
import tempfile
|
|
20
|
+
import time
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
from .. import budget
|
|
24
|
+
from . import project_files
|
|
25
|
+
|
|
26
|
+
CHUNK = 45 * 1024 * 1024
|
|
27
|
+
PKG = Path(__file__).resolve().parents[1] # the agent_testbench package, shipped with the project
|
|
28
|
+
|
|
29
|
+
BOOTSTRAP = r'''
|
|
30
|
+
import glob, json, os, subprocess, sys, tarfile
|
|
31
|
+
parts = sorted(glob.glob("/content/.testbench_bundle.part*"))
|
|
32
|
+
with open("/content/.testbench_bundle.tar.gz", "wb") as out:
|
|
33
|
+
for p in parts:
|
|
34
|
+
with open(p, "rb") as f:
|
|
35
|
+
out.write(f.read())
|
|
36
|
+
os.remove(p)
|
|
37
|
+
safe = {"filter": "data"} if hasattr(tarfile, "data_filter") else {}
|
|
38
|
+
with tarfile.open("/content/.testbench_bundle.tar.gz") as t:
|
|
39
|
+
for m in t.getmembers():
|
|
40
|
+
if m.name.startswith("project/"):
|
|
41
|
+
m.name = m.name[len("project/"):]
|
|
42
|
+
if m.name:
|
|
43
|
+
t.extract(m, "/content", **safe)
|
|
44
|
+
elif m.name.startswith("pkg/"):
|
|
45
|
+
m.name = m.name[len("pkg/"):]
|
|
46
|
+
t.extract(m, "/content/.testbench_pkg", **safe)
|
|
47
|
+
sys.path.insert(0, "/content/.testbench_pkg")
|
|
48
|
+
os.environ["TESTBENCH_CACHE"] = "/content/.testbench_cache"
|
|
49
|
+
os.makedirs("/content/.testbench_cache", exist_ok=True)
|
|
50
|
+
from agent_testbench.executor import execute_plan
|
|
51
|
+
run_dir = "/content/.testbench_run/" + RUN_ID
|
|
52
|
+
report = execute_plan(PLAN, project_dir="/content", run_dir=run_dir, workdir="/content", meta=META)
|
|
53
|
+
print("TESTBENCH_DONE " + json.dumps({"status": report["status"], "minutes": report.get("minutes")}), flush=True)
|
|
54
|
+
'''
|
|
55
|
+
|
|
56
|
+
PACK = r'''
|
|
57
|
+
import os, tarfile
|
|
58
|
+
d = "/content/.testbench_run/" + RUN_ID
|
|
59
|
+
with tarfile.open("/content/.testbench_result.tar.gz", "w:gz") as t:
|
|
60
|
+
if os.path.isdir(d):
|
|
61
|
+
t.add(d, arcname=".")
|
|
62
|
+
print("TESTBENCH_PACKED", os.path.getsize("/content/.testbench_result.tar.gz"), flush=True)
|
|
63
|
+
'''
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def cli_args(cfg: dict) -> list[str]:
|
|
67
|
+
c = cfg.get("colab") or {}
|
|
68
|
+
return [c.get("cli", "colab")] + (["--auth", c["auth"]] if c.get("auth") else [])
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def stop_session(session: str, args: list[str]) -> str:
|
|
72
|
+
try:
|
|
73
|
+
subprocess.run([*(args or ["colab"]), "stop", "-s", session], capture_output=True, text=True, timeout=120)
|
|
74
|
+
return f"Colab session {session} stopped"
|
|
75
|
+
except Exception as exc:
|
|
76
|
+
return f"could not stop Colab session {session}: {exc} - stop it with `colab stop -s {session}`"
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class ColabBackend:
|
|
80
|
+
def __init__(self, root: Path, cfg: dict, run_dir: Path, plan: dict, meta: dict):
|
|
81
|
+
self.root, self.cfg, self.run_dir, self.plan, self.meta = Path(root), cfg, Path(run_dir), plan, meta
|
|
82
|
+
self.args = cli_args(cfg)
|
|
83
|
+
self.session = ("testbench-" + run_dir.name)[:60]
|
|
84
|
+
self.started = None
|
|
85
|
+
|
|
86
|
+
def _colab(self, *argv, timeout: float = 900, check: bool = True, input: str | None = None):
|
|
87
|
+
cmd = [*self.args, *argv]
|
|
88
|
+
proc = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout, input=input)
|
|
89
|
+
with (self.run_dir / "colab.log").open("a") as log:
|
|
90
|
+
log.write(f"$ {' '.join(cmd[:4])} ...\n{proc.stdout[-4000:]}{proc.stderr[-4000:]}\n")
|
|
91
|
+
if check and proc.returncode != 0:
|
|
92
|
+
raise RuntimeError(f"`colab {argv[0]}` failed (exit {proc.returncode}): "
|
|
93
|
+
f"{(proc.stderr or proc.stdout).strip().splitlines()[-1:] or ''}. See colab.log in the run folder.")
|
|
94
|
+
return proc
|
|
95
|
+
|
|
96
|
+
def _bundle(self) -> Path:
|
|
97
|
+
path = Path(tempfile.mkdtemp()) / "bundle.tar.gz"
|
|
98
|
+
with tarfile.open(path, "w:gz") as t:
|
|
99
|
+
for f in project_files(self.root, self.cfg["exclude"]):
|
|
100
|
+
t.add(f, arcname="project/" + str(f.relative_to(self.root)))
|
|
101
|
+
t.add(PKG, arcname="pkg/agent_testbench", filter=lambda m: None if "__pycache__" in m.name else m)
|
|
102
|
+
return path
|
|
103
|
+
|
|
104
|
+
def run(self):
|
|
105
|
+
from .. import runs, report as R
|
|
106
|
+
if not shutil.which(self.args[0]):
|
|
107
|
+
raise RuntimeError("The Colab CLI is not installed: `uv tool install google-colab-cli`, then log in once "
|
|
108
|
+
"with `colab new` (or set colab.auth: adc in testbench.yaml).")
|
|
109
|
+
gpu = (self.plan.get("gpu") or "").upper()
|
|
110
|
+
runs.write_state(self.run_dir, colab_session=self.session, colab_cli_args=self.args)
|
|
111
|
+
print(f"[testbench] allocating Colab session {self.session} ({gpu or 'CPU'})", flush=True)
|
|
112
|
+
self._colab("new", "-s", self.session, *(["--gpu", gpu] if gpu else []), timeout=900)
|
|
113
|
+
self.started = time.time()
|
|
114
|
+
try:
|
|
115
|
+
pkgs = ["nbformat>=5.9", "nbclient>=0.10", "ipykernel>=6.29", "pyyaml>=6", *self.cfg["colab"].get("pip", [])]
|
|
116
|
+
self._colab("install", "-s", self.session, *pkgs, timeout=1200)
|
|
117
|
+
if self.cfg["colab"].get("requirements"):
|
|
118
|
+
self._colab("install", "-s", self.session, "-r", str(self.root / self.cfg["colab"]["requirements"]), timeout=1800)
|
|
119
|
+
bundle = self._bundle()
|
|
120
|
+
data = bundle.read_bytes()
|
|
121
|
+
for i in range(0, max(1, len(data)), CHUNK):
|
|
122
|
+
part = bundle.with_name(f"part{i // CHUNK:03d}")
|
|
123
|
+
part.write_bytes(data[i:i + CHUNK])
|
|
124
|
+
self._colab("upload", "-s", self.session, str(part), f"/content/.testbench_bundle.part{i // CHUNK:03d}", timeout=900)
|
|
125
|
+
minutes = sum(s["max_minutes"] for s in self.plan["steps"])
|
|
126
|
+
script = (f"RUN_ID = {self.run_dir.name!r}\nPLAN = json.loads({json.dumps(self.plan)!r})\n"
|
|
127
|
+
f"META = json.loads({json.dumps(self.meta)!r})\n")
|
|
128
|
+
boot = Path(tempfile.mkdtemp()) / "testbench_boot.py"
|
|
129
|
+
boot.write_text("import json\n" + script + BOOTSTRAP)
|
|
130
|
+
print(f"[testbench] running the plan on Colab (up to {minutes:.0f} min)", flush=True)
|
|
131
|
+
proc = self._colab("exec", "-s", self.session, "-f", str(boot), "--timeout", str(int(minutes * 60 + 120)),
|
|
132
|
+
timeout=minutes * 60 + 600, check=False)
|
|
133
|
+
(self.run_dir / "colab_exec.log").write_text(proc.stdout + proc.stderr)
|
|
134
|
+
pack = Path(tempfile.mkdtemp()) / "testbench_pack.py"
|
|
135
|
+
pack.write_text(f"RUN_ID = {self.run_dir.name!r}\n" + PACK)
|
|
136
|
+
self._colab("exec", "-s", self.session, "-f", str(pack), "--timeout", "300", timeout=600)
|
|
137
|
+
local = Path(tempfile.mkdtemp()) / "result.tar.gz"
|
|
138
|
+
self._colab("download", "-s", self.session, "/content/.testbench_result.tar.gz", str(local), timeout=1800)
|
|
139
|
+
with tarfile.open(local) as t:
|
|
140
|
+
t.extractall(self.run_dir, **({"filter": "data"} if hasattr(tarfile, "data_filter") else {}))
|
|
141
|
+
finally:
|
|
142
|
+
print(f"[testbench] {stop_session(self.session, self.args)}", flush=True)
|
|
143
|
+
rep = R.load(self.run_dir)
|
|
144
|
+
if not rep:
|
|
145
|
+
log = (self.run_dir / "colab_exec.log")
|
|
146
|
+
tail = "\n".join(log.read_text(errors="replace").strip().splitlines()[-20:]) if log.exists() else ""
|
|
147
|
+
rep = {**self.meta, "steps": [], "started": self.started, "status": "error",
|
|
148
|
+
"stopped_because": "the plan never started on the Colab VM",
|
|
149
|
+
"launch_error": "The VM could not start the plan. Its last output:\n\n```\n" + tail + "\n```"}
|
|
150
|
+
return rep, self.cost(rep)
|
|
151
|
+
|
|
152
|
+
def cancel(self) -> str:
|
|
153
|
+
return stop_session(self.session, self.args)
|
|
154
|
+
|
|
155
|
+
def cost(self, rep: dict) -> float:
|
|
156
|
+
minutes = (time.time() - self.started) / 60 if self.started else 0.0
|
|
157
|
+
return budget.rate_per_minute(self.cfg, self.plan) * minutes
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""Run on this machine: free, immediate, and the first rung of every ladder."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
from .. import executor
|
|
7
|
+
from . import copy_project
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class LocalBackend:
|
|
11
|
+
def __init__(self, root: Path, cfg: dict, run_dir: Path, plan: dict, meta: dict):
|
|
12
|
+
self.root, self.cfg, self.run_dir, self.plan, self.meta = Path(root), cfg, Path(run_dir), plan, meta
|
|
13
|
+
|
|
14
|
+
def run(self):
|
|
15
|
+
workdir = self.root
|
|
16
|
+
if self.plan.get("workdir") == "copy":
|
|
17
|
+
workdir = copy_project(self.root, self.run_dir / "work", self.cfg["exclude"])
|
|
18
|
+
rep = executor.execute_plan(self.plan, project_dir=self.root, run_dir=self.run_dir, workdir=workdir, meta=self.meta)
|
|
19
|
+
return rep, 0.0
|
|
20
|
+
|
|
21
|
+
def cancel(self) -> str:
|
|
22
|
+
return ""
|
|
23
|
+
|
|
24
|
+
def cost(self, rep: dict) -> float:
|
|
25
|
+
return 0.0
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
"""Run on Modal: one deployed app per project, one function per plan, results on a Modal volume.
|
|
2
|
+
|
|
3
|
+
The app is generated from testbench.yaml into .testbench/modal_app.py (readable, not hand-edited), deployed, and
|
|
4
|
+
the plan's function is spawned, so the run keeps going if this machine sleeps or loses its network. Lessons
|
|
5
|
+
built in:
|
|
6
|
+
- the function's timeout is the plan's budget, so the worst case holds on Modal's side;
|
|
7
|
+
- CPU and memory are always explicit (a GPU function without cpu= gets one core, and the GPU waits on it);
|
|
8
|
+
- no retries and a 2-second scaledown, so a finished or failed run stops billing at once;
|
|
9
|
+
- local files are added as the image's last steps, so editing code does not rebuild the image.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import subprocess
|
|
15
|
+
import sys
|
|
16
|
+
import textwrap
|
|
17
|
+
import threading
|
|
18
|
+
import time
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from .. import budget
|
|
22
|
+
|
|
23
|
+
APP_TEMPLATE = '''\
|
|
24
|
+
# Generated by agent-testbench from testbench.yaml - edit testbench.yaml, not this file. Regenerated on every Modal run.
|
|
25
|
+
import json
|
|
26
|
+
import threading
|
|
27
|
+
import time
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
|
|
30
|
+
import modal
|
|
31
|
+
|
|
32
|
+
app = modal.App({app_name!r})
|
|
33
|
+
volume = modal.Volume.from_name({volume_name!r}, create_if_missing=True)
|
|
34
|
+
image = (
|
|
35
|
+
modal.Image.debian_slim(python_version={python!r})
|
|
36
|
+
{image_steps}
|
|
37
|
+
.add_local_python_source("agent_testbench")
|
|
38
|
+
.add_local_dir({root!r}, "/project", ignore={ignore!r})
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _run(run_id: str, plan: dict, meta: dict) -> dict:
|
|
43
|
+
import os
|
|
44
|
+
import shutil
|
|
45
|
+
from agent_testbench.executor import execute_plan
|
|
46
|
+
|
|
47
|
+
work = Path("/work/project")
|
|
48
|
+
shutil.copytree("/project", work, dirs_exist_ok=True, symlinks=True)
|
|
49
|
+
if not Path("/content").exists(): # Colab notebooks expect to live in /content
|
|
50
|
+
Path("/content").symlink_to(work)
|
|
51
|
+
os.environ["TESTBENCH_CACHE"] = "/testbench/cache"
|
|
52
|
+
Path("/testbench/cache").mkdir(parents=True, exist_ok=True)
|
|
53
|
+
run_dir = Path("/testbench/runs") / run_id
|
|
54
|
+
stop = threading.Event()
|
|
55
|
+
|
|
56
|
+
def keep_committing(): # live logs reach the volume about once a minute
|
|
57
|
+
while not stop.wait(60):
|
|
58
|
+
try:
|
|
59
|
+
volume.commit()
|
|
60
|
+
except Exception:
|
|
61
|
+
pass
|
|
62
|
+
|
|
63
|
+
threading.Thread(target=keep_committing, daemon=True).start()
|
|
64
|
+
try:
|
|
65
|
+
report = execute_plan(plan, project_dir=work, run_dir=run_dir, workdir=work, meta=meta, on_progress=volume.commit)
|
|
66
|
+
finally:
|
|
67
|
+
stop.set()
|
|
68
|
+
volume.commit()
|
|
69
|
+
return {{"status": report["status"], "minutes": report.get("minutes")}}
|
|
70
|
+
{functions}
|
|
71
|
+
'''
|
|
72
|
+
|
|
73
|
+
FUNCTION_TEMPLATE = '''
|
|
74
|
+
|
|
75
|
+
@app.function(image=image, gpu={gpu!r}, cpu={cpu!r}, memory={memory!r}, timeout={timeout!r}, retries=0,
|
|
76
|
+
scaledown_window=2, volumes={{"/testbench": volume}}, secrets=[{secrets}])
|
|
77
|
+
def {fn}(run_id: str, plan: dict, meta: dict) -> dict:
|
|
78
|
+
return _run(run_id, plan, meta)
|
|
79
|
+
'''
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def app_name(cfg: dict) -> str:
|
|
83
|
+
return f"testbench-{cfg['project']}"
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def fn_name(plan_name: str) -> str:
|
|
87
|
+
return "plan_" + "".join(c if c.isalnum() else "_" for c in plan_name)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def generate(cfg: dict) -> str:
|
|
91
|
+
m = cfg["modal"]
|
|
92
|
+
steps = []
|
|
93
|
+
if m.get("apt"):
|
|
94
|
+
steps.append(f" .apt_install({', '.join(repr(x) for x in m['apt'])})")
|
|
95
|
+
base = ["nbformat>=5.9", "nbclient>=0.10", "ipykernel>=6.29", "pyyaml>=6"]
|
|
96
|
+
uses_web = any(s["kind"] == "web" for p in cfg["plans"].values() if p["backend"] == "modal" for s in p["steps"])
|
|
97
|
+
if uses_web:
|
|
98
|
+
base.append("playwright>=1.45")
|
|
99
|
+
steps.append(f" .pip_install({', '.join(repr(x) for x in base + list(m.get('pip') or []))})")
|
|
100
|
+
if m.get("requirements"):
|
|
101
|
+
req = Path(cfg["root"]) / m["requirements"]
|
|
102
|
+
steps.append(f" .pip_install_from_requirements({str(req)!r})")
|
|
103
|
+
if uses_web:
|
|
104
|
+
steps.append(' .run_commands("playwright install --with-deps chromium")')
|
|
105
|
+
for cmd in m.get("run_commands") or []:
|
|
106
|
+
steps.append(f" .run_commands({cmd!r})")
|
|
107
|
+
ignore = []
|
|
108
|
+
for p in list(cfg["exclude"]) + [".testbench"]:
|
|
109
|
+
ignore += [p] if "/" in p else [p, f"**/{p}"]
|
|
110
|
+
functions = []
|
|
111
|
+
for name, plan in cfg["plans"].items():
|
|
112
|
+
if plan["backend"] != "modal":
|
|
113
|
+
continue
|
|
114
|
+
minutes = sum(s["max_minutes"] for s in plan["steps"])
|
|
115
|
+
secrets = sorted({s for st in plan["steps"] for s in st["secrets"]})
|
|
116
|
+
functions.append(FUNCTION_TEMPLATE.format(
|
|
117
|
+
gpu=plan.get("gpu"), cpu=float(plan.get("cpu", m["cpu"])), memory=int(float(plan.get("memory_gb", m["memory_gb"])) * 1024),
|
|
118
|
+
timeout=int((minutes + 3) * 60), fn=fn_name(name),
|
|
119
|
+
secrets=", ".join(f"modal.Secret.from_name({s!r})" for s in secrets)))
|
|
120
|
+
return APP_TEMPLATE.format(app_name=app_name(cfg), volume_name=app_name(cfg), python=str(m["python"]),
|
|
121
|
+
image_steps="\n".join(steps), root=str(cfg["root"]), ignore=ignore,
|
|
122
|
+
functions="".join(functions))
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def cancel_call(call_id: str) -> str:
|
|
126
|
+
import modal
|
|
127
|
+
try:
|
|
128
|
+
modal.FunctionCall.from_id(call_id).cancel(terminate_containers=True)
|
|
129
|
+
return f"Modal call {call_id} cancelled"
|
|
130
|
+
except Exception as exc:
|
|
131
|
+
return f"could not cancel Modal call {call_id}: {exc}"
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
class ModalBackend:
|
|
135
|
+
def __init__(self, root: Path, cfg: dict, run_dir: Path, plan: dict, meta: dict):
|
|
136
|
+
self.root, self.cfg, self.run_dir, self.plan, self.meta = Path(root), cfg, Path(run_dir), plan, meta
|
|
137
|
+
self.call = None
|
|
138
|
+
self.remote_started = None
|
|
139
|
+
self._seen: dict[str, tuple] = {}
|
|
140
|
+
|
|
141
|
+
def _log(self, text: str) -> None:
|
|
142
|
+
print(f"[testbench] {text}", flush=True)
|
|
143
|
+
|
|
144
|
+
def deploy(self) -> None:
|
|
145
|
+
app_file = self.root / ".testbench" / "modal_app.py"
|
|
146
|
+
app_file.parent.mkdir(parents=True, exist_ok=True)
|
|
147
|
+
app_file.write_text(generate(self.cfg))
|
|
148
|
+
self._log(f"deploying {app_name(self.cfg)} from {app_file}")
|
|
149
|
+
proc = subprocess.run([sys.executable, "-m", "modal", "deploy", str(app_file)], cwd=self.root,
|
|
150
|
+
capture_output=True, text=True, timeout=1800)
|
|
151
|
+
(self.run_dir / "modal_deploy.log").write_text(proc.stdout + proc.stderr)
|
|
152
|
+
if proc.returncode != 0:
|
|
153
|
+
tail = "\n".join((proc.stdout + proc.stderr).strip().splitlines()[-25:])
|
|
154
|
+
raise RuntimeError(f"`modal deploy` failed (is Modal set up? run `modal setup`):\n{tail}")
|
|
155
|
+
|
|
156
|
+
def run(self):
|
|
157
|
+
import modal
|
|
158
|
+
from .. import runs
|
|
159
|
+
self.deploy()
|
|
160
|
+
fn = modal.Function.from_name(app_name(self.cfg), fn_name(self.meta["plan"]))
|
|
161
|
+
self.call = fn.spawn(self.run_dir.name, self.plan, self.meta)
|
|
162
|
+
self.remote_started = time.time()
|
|
163
|
+
runs.write_state(self.run_dir, modal_call_id=self.call.object_id, modal_app=app_name(self.cfg))
|
|
164
|
+
self._log(f"spawned Modal call {self.call.object_id}")
|
|
165
|
+
volume = modal.Volume.from_name(app_name(self.cfg))
|
|
166
|
+
result = None
|
|
167
|
+
while True:
|
|
168
|
+
try:
|
|
169
|
+
result = self.call.get(timeout=20)
|
|
170
|
+
break
|
|
171
|
+
except (TimeoutError, modal.exception.TimeoutError):
|
|
172
|
+
self._mirror(volume, light=True)
|
|
173
|
+
except modal.exception.FunctionTimeoutError:
|
|
174
|
+
result = {"status": "timeout"}
|
|
175
|
+
break
|
|
176
|
+
self._mirror(volume, light=False)
|
|
177
|
+
from .. import report as R
|
|
178
|
+
rep = R.load(self.run_dir) or {**self.meta, "steps": [], "started": self.remote_started, "status": "error"}
|
|
179
|
+
if result and result.get("status") == "timeout":
|
|
180
|
+
rep.update(status="failed", stopped_because="the plan ran past its total max_minutes on Modal")
|
|
181
|
+
return rep, self.cost(rep)
|
|
182
|
+
|
|
183
|
+
def _mirror(self, volume, light: bool) -> None:
|
|
184
|
+
"""Copy the run's files from the volume: progress and logs while it runs, everything at the end."""
|
|
185
|
+
prefix = f"runs/{self.run_dir.name}"
|
|
186
|
+
try:
|
|
187
|
+
entries = volume.listdir(prefix, recursive=True)
|
|
188
|
+
except Exception:
|
|
189
|
+
return
|
|
190
|
+
for e in entries:
|
|
191
|
+
if getattr(e.type, "name", str(e.type)) != "FILE":
|
|
192
|
+
continue
|
|
193
|
+
rel = Path(e.path).relative_to(prefix)
|
|
194
|
+
if light and not (rel.name in ("progress.jsonl", "report.json", "log.txt", "result.json")):
|
|
195
|
+
continue
|
|
196
|
+
stamp = (getattr(e, "size", None), getattr(e, "mtime", None))
|
|
197
|
+
if self._seen.get(e.path) == stamp and stamp != (None, None):
|
|
198
|
+
continue # unchanged since the last copy
|
|
199
|
+
self._seen[e.path] = stamp
|
|
200
|
+
target = self.run_dir / rel
|
|
201
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
202
|
+
try:
|
|
203
|
+
target.write_bytes(b"".join(volume.read_file(e.path)))
|
|
204
|
+
except Exception:
|
|
205
|
+
pass
|
|
206
|
+
|
|
207
|
+
def cancel(self) -> str:
|
|
208
|
+
return cancel_call(self.call.object_id) if self.call else ""
|
|
209
|
+
|
|
210
|
+
def cost(self, rep: dict) -> float:
|
|
211
|
+
minutes = (time.time() - self.remote_started) / 60 if self.remote_started else 0.0
|
|
212
|
+
return budget.rate_per_minute(self.cfg, self.plan) * (minutes + 0.5)
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""Money: what a plan can cost at worst, what runs have cost, and whether a launch is allowed.
|
|
2
|
+
|
|
3
|
+
The worst case of a plan is its hardware's rate times the sum of its steps' max_minutes. Remote steps are
|
|
4
|
+
also stopped on the remote side at that limit (a Modal function timeout, a Colab exec timeout), so the
|
|
5
|
+
estimate is a bound, not a hope - it holds even if this machine goes offline mid-run.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import datetime as dt
|
|
10
|
+
import json
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from . import config as C
|
|
14
|
+
|
|
15
|
+
# Modal, US dollars per second (modal.com/pricing, checked 2026-10). Override any of them under `pricing:`.
|
|
16
|
+
MODAL_GPU_PER_S = {"B300": 0.001972, "B200": 0.001736, "H200": 0.001261, "H100": 0.001097, "RTX-PRO-6000": 0.000842,
|
|
17
|
+
"A100-80GB": 0.000694, "A100-40GB": 0.000583, "A100": 0.000583, "L40S": 0.000542, "A10": 0.000306,
|
|
18
|
+
"A10G": 0.000306, "L4": 0.000222, "T4": 0.000164}
|
|
19
|
+
MODAL_CPU_CORE_PER_S = 0.0000131
|
|
20
|
+
MODAL_MEM_GIB_PER_S = 0.00000222
|
|
21
|
+
# Colab bills compute units. These per-hour rates are approximate and vary by account and region; `colab usage`
|
|
22
|
+
# shows your actual rate - put it under `pricing: colab_units_per_hour:`. Pay-as-you-go is ~$10 per 100 units.
|
|
23
|
+
COLAB_UNITS_PER_HOUR = {"CPU": 0.1, "T4": 1.8, "L4": 2.0, "G4": 4.0, "A100": 5.4, "H100": 9.0}
|
|
24
|
+
COLAB_USD_PER_UNIT = 0.10
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def rate_per_minute(cfg: dict, plan: dict) -> float:
|
|
28
|
+
"""US dollars per minute for the plan's hardware; 0 for local plans."""
|
|
29
|
+
prices = cfg.get("pricing") or {}
|
|
30
|
+
backend = plan["backend"]
|
|
31
|
+
if backend == "local":
|
|
32
|
+
return 0.0
|
|
33
|
+
gpu = (plan.get("gpu") or "").split(":")[0].upper()
|
|
34
|
+
count = int(plan["gpu"].split(":")[1]) if plan.get("gpu") and ":" in plan["gpu"] else 1
|
|
35
|
+
if backend == "modal":
|
|
36
|
+
gpu_rates = {**MODAL_GPU_PER_S, **(prices.get("modal_gpu_per_s") or {})}
|
|
37
|
+
cpu = float(plan.get("cpu") or cfg["modal"]["cpu"])
|
|
38
|
+
mem = float(plan.get("memory_gb") or cfg["modal"]["memory_gb"])
|
|
39
|
+
per_s = (gpu_rates.get(gpu, 0.0) * count if gpu else 0.0) \
|
|
40
|
+
+ cpu * float(prices.get("modal_cpu_core_per_s", MODAL_CPU_CORE_PER_S)) \
|
|
41
|
+
+ mem * float(prices.get("modal_mem_gib_per_s", MODAL_MEM_GIB_PER_S))
|
|
42
|
+
return per_s * 60
|
|
43
|
+
units = {**COLAB_UNITS_PER_HOUR, **(prices.get("colab_units_per_hour") or {})}
|
|
44
|
+
return units.get(gpu or "CPU", 0.0) * float(prices.get("colab_usd_per_unit", COLAB_USD_PER_UNIT)) / 60
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def estimate(cfg: dict, hardware: dict, minutes: float, label: str) -> dict:
|
|
48
|
+
"""Worst case for `minutes` on `hardware` ({backend, gpu, cpu?, memory_gb?}); a remote machine also pays for
|
|
49
|
+
starting up and installing, so a few minutes are added."""
|
|
50
|
+
rate = rate_per_minute(cfg, hardware)
|
|
51
|
+
overhead = 0 if hardware["backend"] == "local" else 5
|
|
52
|
+
return {"what": label, "backend": hardware["backend"], "gpu": hardware.get("gpu"), "usd_per_minute": rate,
|
|
53
|
+
"max_minutes": float(minutes), "worst_case_usd": round(rate * (float(minutes) + overhead), 4)}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def worst_case(cfg: dict, plan_name: str) -> dict:
|
|
57
|
+
plan = cfg["plans"][plan_name]
|
|
58
|
+
return {**estimate(cfg, plan, C.plan_minutes(plan), f"plan {plan_name}"), "plan": plan_name}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def ledger_path(root: Path) -> Path:
|
|
62
|
+
return Path(root) / ".testbench" / "ledger.jsonl"
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def record(root: Path, entry: dict) -> None:
|
|
66
|
+
p = ledger_path(root)
|
|
67
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
68
|
+
with p.open("a") as f:
|
|
69
|
+
f.write(json.dumps({"time": dt.datetime.now().isoformat(timespec="seconds"), **entry}) + "\n")
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def entries(root: Path) -> list[dict]:
|
|
73
|
+
p = ledger_path(root)
|
|
74
|
+
if not p.exists():
|
|
75
|
+
return []
|
|
76
|
+
return [json.loads(line) for line in p.read_text().splitlines() if line.strip()]
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def spent_today(root: Path) -> float:
|
|
80
|
+
"""Dollars committed today: finished runs at their measured cost, unfinished ones at their worst case."""
|
|
81
|
+
today = dt.date.today().isoformat()
|
|
82
|
+
runs: dict[str, float] = {}
|
|
83
|
+
for e in entries(root):
|
|
84
|
+
if not e["time"].startswith(today):
|
|
85
|
+
continue
|
|
86
|
+
if e["event"] == "launch":
|
|
87
|
+
runs[e["run"]] = e["worst_case_usd"]
|
|
88
|
+
elif e["event"] == "finish":
|
|
89
|
+
runs[e["run"]] = e["usd"]
|
|
90
|
+
return round(sum(runs.values()), 4)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def check(cfg: dict, est: dict, approve: bool, how_to_lower: str) -> tuple[bool, str, dict]:
|
|
94
|
+
"""(allowed, message, estimate). Over max_run_usd or the day's budget: refused. Over ask_above_usd: needs
|
|
95
|
+
approve=True, which an agent may only pass after a person said yes."""
|
|
96
|
+
b = cfg["budget"]
|
|
97
|
+
cost, what = est["worst_case_usd"], est["what"]
|
|
98
|
+
if cost > b["max_run_usd"]:
|
|
99
|
+
return False, (f"Refused: {what} could cost up to ${cost:.2f} ({est['max_minutes']:.0f} min at "
|
|
100
|
+
f"${est['usd_per_minute']:.4f}/min), over budget.max_run_usd ${b['max_run_usd']:.2f}. "
|
|
101
|
+
f"{how_to_lower}, use a cheaper GPU, or have a person raise the budget."), est
|
|
102
|
+
today = spent_today(Path(cfg["root"]))
|
|
103
|
+
if today + cost > b["max_day_usd"]:
|
|
104
|
+
return False, (f"Refused: ${today:.2f} already committed today; {what} could add ${cost:.2f}, over "
|
|
105
|
+
f"budget.max_day_usd ${b['max_day_usd']:.2f}."), est
|
|
106
|
+
if cost > b["ask_above_usd"] and not approve:
|
|
107
|
+
return False, (f"Needs approval: {what} could cost up to ${cost:.2f} (above budget.ask_above_usd "
|
|
108
|
+
f"${b['ask_above_usd']:.2f}). Ask a person; if they agree, rerun with --approve."), est
|
|
109
|
+
return True, f"Within budget: up to ${cost:.2f} (today so far ${today:.2f} of ${b['max_day_usd']:.2f}).", est
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def check_launch(cfg: dict, plan_name: str, approve: bool) -> tuple[bool, str, dict]:
|
|
113
|
+
return check(cfg, worst_case(cfg, plan_name), approve, "Lower the steps' max_minutes")
|