lab-kit-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lab_kit/__init__.py +3 -0
- lab_kit/__main__.py +3 -0
- lab_kit/_data/agents/reporter.md +30 -0
- lab_kit/_data/agents/reviewer.md +36 -0
- lab_kit/_data/agents/runner.md +41 -0
- lab_kit/_data/agents/scout.md +28 -0
- lab_kit/_data/method/DISCIPLINE.md +110 -0
- lab_kit/_data/method/LADDER.md +72 -0
- lab_kit/_data/skills/experiment/SKILL.md +80 -0
- lab_kit/_data/skills/plan-mission/SKILL.md +78 -0
- lab_kit/_data/skills/review/SKILL.md +75 -0
- lab_kit/_data/skills/run-mission/SKILL.md +83 -0
- lab_kit/_data/skills/set-up-lab/SKILL.md +81 -0
- lab_kit/checks/__init__.py +1 -0
- lab_kit/checks/base.py +41 -0
- lab_kit/checks/files.py +312 -0
- lab_kit/checks/locks.py +134 -0
- lab_kit/checks/results.py +172 -0
- lab_kit/checks/run.py +63 -0
- lab_kit/cli.py +174 -0
- lab_kit/commands/__init__.py +1 -0
- lab_kit/commands/experiments.py +89 -0
- lab_kit/commands/runs.py +105 -0
- lab_kit/commands/setup.py +126 -0
- lab_kit/commands/status.py +26 -0
- lab_kit/data.py +49 -0
- lab_kit/errors.py +2 -0
- lab_kit/frozen.py +48 -0
- lab_kit/gitlog.py +55 -0
- lab_kit/lab.py +135 -0
- lab_kit/ops.py +94 -0
- lab_kit/protocol.py +100 -0
- lab_kit/records.py +245 -0
- lab_kit/rederive.py +54 -0
- lab_kit/supervise.py +80 -0
- lab_kit_cli-0.1.0.dist-info/METADATA +92 -0
- lab_kit_cli-0.1.0.dist-info/RECORD +40 -0
- lab_kit_cli-0.1.0.dist-info/WHEEL +4 -0
- lab_kit_cli-0.1.0.dist-info/entry_points.txt +2 -0
- lab_kit_cli-0.1.0.dist-info/licenses/LICENSE +21 -0
lab_kit/records.py
ADDED
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
"""The lab files lab-kit writes and reads: lock records, run records, manifests and scorecards."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime as _dt
|
|
6
|
+
import hashlib
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
import yaml
|
|
14
|
+
|
|
15
|
+
from .errors import LabError
|
|
16
|
+
from .lab import Lab
|
|
17
|
+
|
|
18
|
+
LOCK = "lock.json"
|
|
19
|
+
RUN = "run.json"
|
|
20
|
+
MANIFEST = "MANIFEST.sha256"
|
|
21
|
+
SCORE = "score.yaml"
|
|
22
|
+
SPEND = "spend.json" # written by a spending run's own command into out/
|
|
23
|
+
LOCK_KEYS = ("protocol", "path", "sha256", "locked")
|
|
24
|
+
RUN_KEYS = ("protocol", "lock", "config", "mission", "spend", "command", "started", "ended", "exit", "pid",
|
|
25
|
+
"spent")
|
|
26
|
+
PREDICTION_VERDICTS = ("HIT", "MISS", "INDETERMINATE")
|
|
27
|
+
RULE_VERDICTS = ("FIRED", "NOT FIRED", "INDETERMINATE")
|
|
28
|
+
REVIEW_VERDICTS = ("pass", "fail")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def now() -> str:
|
|
32
|
+
"""The current moment in UTC, to the second, as ISO 8601."""
|
|
33
|
+
return _dt.datetime.now(_dt.timezone.utc).isoformat(timespec="seconds")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def script_env() -> dict[str, str]:
|
|
37
|
+
"""The environment lab-kit runs a lab's scripts in: the caller's, with no bytecode written.
|
|
38
|
+
|
|
39
|
+
A `__pycache__` folder written into a frozen surface would change it.
|
|
40
|
+
"""
|
|
41
|
+
env = dict(os.environ)
|
|
42
|
+
env["PYTHONDONTWRITEBYTECODE"] = "1"
|
|
43
|
+
return env
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def parse_moment(value: Any, where: str) -> _dt.datetime:
|
|
47
|
+
if not isinstance(value, str):
|
|
48
|
+
raise LabError(f"{where}: expected an ISO 8601 time, not {value!r}")
|
|
49
|
+
try:
|
|
50
|
+
moment = _dt.datetime.fromisoformat(value.replace("Z", "+00:00"))
|
|
51
|
+
except ValueError as exc:
|
|
52
|
+
raise LabError(f"{where}: `{value}` is not an ISO 8601 time") from exc
|
|
53
|
+
if moment.tzinfo is None:
|
|
54
|
+
moment = moment.replace(tzinfo=_dt.timezone.utc)
|
|
55
|
+
return moment
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def read_json(path: Path, where: str) -> dict[str, Any]:
|
|
59
|
+
try:
|
|
60
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
61
|
+
except json.JSONDecodeError as exc:
|
|
62
|
+
raise LabError(f"{where}: not valid JSON: {exc}") from exc
|
|
63
|
+
if not isinstance(data, dict):
|
|
64
|
+
raise LabError(f"{where}: must be a JSON object")
|
|
65
|
+
return data
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def write_json(path: Path, data: dict[str, Any]) -> None:
|
|
69
|
+
tmp = path.with_name(path.name + ".tmp")
|
|
70
|
+
tmp.write_text(json.dumps(data, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
|
|
71
|
+
os.replace(tmp, path)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def read_yaml(path: Path, where: str) -> dict[str, Any]:
|
|
75
|
+
try:
|
|
76
|
+
data = yaml.safe_load(path.read_text(encoding="utf-8"))
|
|
77
|
+
except yaml.YAMLError as exc:
|
|
78
|
+
raise LabError(f"{where}: not valid YAML: {exc}") from exc
|
|
79
|
+
if not isinstance(data, dict):
|
|
80
|
+
raise LabError(f"{where}: must be a mapping")
|
|
81
|
+
return data
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
# ---------------------------------------------------------------- locks
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
@dataclass
|
|
88
|
+
class LockRecord:
|
|
89
|
+
slug: str # the experiment folder's name
|
|
90
|
+
path: str # lock.json from the lab root
|
|
91
|
+
data: dict[str, Any]
|
|
92
|
+
|
|
93
|
+
@property
|
|
94
|
+
def protocol(self) -> str:
|
|
95
|
+
return str(self.data.get("protocol", ""))
|
|
96
|
+
|
|
97
|
+
@property
|
|
98
|
+
def sha256(self) -> str:
|
|
99
|
+
return str(self.data.get("sha256", ""))
|
|
100
|
+
|
|
101
|
+
@property
|
|
102
|
+
def locked(self) -> _dt.datetime:
|
|
103
|
+
return parse_moment(self.data.get("locked"), self.path)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def lock_path(lab: Lab, slug: str) -> Path:
|
|
107
|
+
return lab.experiments / slug / LOCK
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def read_lock(lab: Lab, slug: str) -> LockRecord | None:
|
|
111
|
+
path = lock_path(lab, slug)
|
|
112
|
+
if not path.is_file():
|
|
113
|
+
return None
|
|
114
|
+
where = lab.rel(path)
|
|
115
|
+
data = read_json(path, where)
|
|
116
|
+
missing = [k for k in LOCK_KEYS if k not in data]
|
|
117
|
+
if missing:
|
|
118
|
+
raise LabError(f"{where}: missing {', '.join(missing)}")
|
|
119
|
+
return LockRecord(slug, where, data)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
# ---------------------------------------------------------------- runs
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
@dataclass
|
|
126
|
+
class RunRecord:
|
|
127
|
+
slug: str
|
|
128
|
+
run_id: str
|
|
129
|
+
folder: Path
|
|
130
|
+
data: dict[str, Any]
|
|
131
|
+
|
|
132
|
+
@property
|
|
133
|
+
def path(self) -> str:
|
|
134
|
+
return f"experiments/{self.slug}/runs/{self.run_id}/{RUN}"
|
|
135
|
+
|
|
136
|
+
@property
|
|
137
|
+
def out(self) -> Path:
|
|
138
|
+
return self.folder / "out"
|
|
139
|
+
|
|
140
|
+
@property
|
|
141
|
+
def manifest(self) -> Path:
|
|
142
|
+
return self.folder / MANIFEST
|
|
143
|
+
|
|
144
|
+
@property
|
|
145
|
+
def ended(self) -> bool:
|
|
146
|
+
return self.data.get("ended") is not None
|
|
147
|
+
|
|
148
|
+
@property
|
|
149
|
+
def state(self) -> str:
|
|
150
|
+
"""running, finished, failed or orphaned."""
|
|
151
|
+
if self.ended:
|
|
152
|
+
return "finished" if self.data.get("exit") == 0 else "failed"
|
|
153
|
+
return "running" if _alive(self.data.get("pid")) else "orphaned"
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _alive(pid: Any) -> bool:
|
|
157
|
+
if not isinstance(pid, int) or isinstance(pid, bool) or pid <= 0:
|
|
158
|
+
return False
|
|
159
|
+
try:
|
|
160
|
+
os.kill(pid, 0)
|
|
161
|
+
except ProcessLookupError:
|
|
162
|
+
return False
|
|
163
|
+
except PermissionError:
|
|
164
|
+
return True
|
|
165
|
+
except OverflowError:
|
|
166
|
+
return False
|
|
167
|
+
cmdline = Path(f"/proc/{pid}/cmdline") # where it exists, tell a reused pid from the supervisor
|
|
168
|
+
if cmdline.is_file():
|
|
169
|
+
try:
|
|
170
|
+
return b"lab_kit.supervise" in cmdline.read_bytes()
|
|
171
|
+
except OSError:
|
|
172
|
+
return True
|
|
173
|
+
return True
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def runs(lab: Lab, slug: str | None = None) -> list[RunRecord]:
|
|
177
|
+
"""Every run with a run.json, oldest first within each experiment."""
|
|
178
|
+
out: list[RunRecord] = []
|
|
179
|
+
for name in [slug] if slug else lab.experiment_slugs():
|
|
180
|
+
folder = lab.experiments / name / "runs"
|
|
181
|
+
if not folder.is_dir():
|
|
182
|
+
continue
|
|
183
|
+
for run_dir in sorted(p for p in folder.iterdir() if p.is_dir()):
|
|
184
|
+
path = run_dir / RUN
|
|
185
|
+
if not path.is_file():
|
|
186
|
+
continue
|
|
187
|
+
out.append(RunRecord(name, run_dir.name, run_dir, read_json(path, lab.rel(path))))
|
|
188
|
+
return out
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def runs_without_record(lab: Lab) -> list[str]:
|
|
192
|
+
"""Run folders holding no run.json, from the lab root."""
|
|
193
|
+
out = []
|
|
194
|
+
for name in lab.experiment_slugs():
|
|
195
|
+
folder = lab.experiments / name / "runs"
|
|
196
|
+
if folder.is_dir():
|
|
197
|
+
out.extend(lab.rel(p) for p in sorted(folder.iterdir()) if p.is_dir() and not (p / RUN).is_file())
|
|
198
|
+
return out
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
# ---------------------------------------------------------------- manifests
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def sha256_file(path: Path) -> str:
|
|
205
|
+
digest = hashlib.sha256()
|
|
206
|
+
with path.open("rb") as handle:
|
|
207
|
+
for chunk in iter(lambda: handle.read(1 << 16), b""):
|
|
208
|
+
digest.update(chunk)
|
|
209
|
+
return digest.hexdigest()
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def tree_hashes(folder: Path, base: Path) -> dict[str, str]:
|
|
213
|
+
"""Every file under `folder`, by its path from `base`, with its sha256."""
|
|
214
|
+
if folder.is_file():
|
|
215
|
+
return {folder.relative_to(base).as_posix(): sha256_file(folder)}
|
|
216
|
+
if not folder.is_dir():
|
|
217
|
+
return {}
|
|
218
|
+
return {p.relative_to(base).as_posix(): sha256_file(p) for p in sorted(folder.rglob("*")) if p.is_file()}
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def format_hashes(hashes: dict[str, str]) -> str:
|
|
222
|
+
return "".join(f"{digest} {path}\n" for path, digest in sorted(hashes.items()))
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def parse_hashes(text: str, where: str) -> dict[str, str]:
|
|
226
|
+
out: dict[str, str] = {}
|
|
227
|
+
for number, line in enumerate(text.splitlines(), 1):
|
|
228
|
+
if not line.strip():
|
|
229
|
+
continue
|
|
230
|
+
digest, sep, path = line.partition(" ")
|
|
231
|
+
if not sep or len(digest) != 64 or not path:
|
|
232
|
+
raise LabError(f"{where}:{number}: expected `<sha256> <path>`")
|
|
233
|
+
out[path] = digest
|
|
234
|
+
return out
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def write_manifest(run: RunRecord) -> None:
|
|
238
|
+
run.manifest.write_text(format_hashes(tree_hashes(run.out, run.out)), encoding="utf-8")
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
# ---------------------------------------------------------------- scorecards
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def score_path(lab: Lab, slug: str) -> Path:
|
|
245
|
+
return lab.experiments / slug / SCORE
|
lab_kit/rederive.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Run a result's re-derive command from the lab root and compare what it prints with its number."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import subprocess
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
from folio.documents import Document
|
|
9
|
+
|
|
10
|
+
from .errors import LabError
|
|
11
|
+
from .lab import Lab
|
|
12
|
+
from .records import script_env
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass
|
|
16
|
+
class Outcome:
|
|
17
|
+
ok: bool
|
|
18
|
+
expected: str
|
|
19
|
+
printed: str
|
|
20
|
+
detail: str # why it failed, or "" when it passed
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def number_of(doc: Document) -> str:
|
|
24
|
+
return str(doc.meta.get("number", "")).strip()
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def run(lab: Lab, doc: Document) -> Outcome:
|
|
28
|
+
command = doc.meta.get("rederive")
|
|
29
|
+
expected = number_of(doc)
|
|
30
|
+
if not isinstance(command, str) or not command.strip():
|
|
31
|
+
return Outcome(False, expected, "", "has no `rederive` command")
|
|
32
|
+
try:
|
|
33
|
+
done = subprocess.run(command, shell=True, cwd=lab.root, capture_output=True, text=True,
|
|
34
|
+
timeout=lab.settings.rederive_timeout, check=False, env=script_env())
|
|
35
|
+
except subprocess.TimeoutExpired:
|
|
36
|
+
return Outcome(False, expected, "", f"`{command}` ran past the {lab.settings.rederive_timeout}s timeout")
|
|
37
|
+
printed = done.stdout.strip()
|
|
38
|
+
if done.returncode != 0:
|
|
39
|
+
tail = done.stderr.strip().splitlines()[-1:] or [""]
|
|
40
|
+
return Outcome(False, expected, printed, f"`{command}` exited {done.returncode}: {tail[0]}".rstrip(": "))
|
|
41
|
+
if printed != expected:
|
|
42
|
+
return Outcome(False, expected, printed, f"`{command}` printed `{printed}`, not the number `{expected}`")
|
|
43
|
+
return Outcome(True, expected, printed, "")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def result_doc(lab: Lab, ident: str) -> Document:
|
|
47
|
+
docs = lab.library().find(ident, "result")
|
|
48
|
+
if not docs:
|
|
49
|
+
raise LabError(f"no result `{ident}` in the library")
|
|
50
|
+
return docs[0]
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def live_results(lab: Lab) -> list[Document]:
|
|
54
|
+
return [d for d in lab.library().documents if d.is_a("result") and d.status == "live"]
|
lab_kit/supervise.py
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""The background half of `lab-kit run`: run the lab's command, then seal its evidence.
|
|
2
|
+
|
|
3
|
+
`lab-kit run` starts this module as a detached process with the run's folder,
|
|
4
|
+
and writes its pid into `run.json`. This process waits for that, runs the command from the lab root with its output in
|
|
5
|
+
`run.log`, and when the command exits writes `MANIFEST.sha256` for `out/`,
|
|
6
|
+
the exit code and the end time into `run.json`.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
import os
|
|
13
|
+
import subprocess
|
|
14
|
+
import sys
|
|
15
|
+
import time
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
from . import records
|
|
19
|
+
|
|
20
|
+
ENV_RUN = "LAB_RUN_DIR"
|
|
21
|
+
ENV_OUT = "LAB_OUT"
|
|
22
|
+
ENV_WORK = "LAB_WORK"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _spent(out: Path) -> float | None:
|
|
26
|
+
path = out / records.SPEND
|
|
27
|
+
if not path.is_file():
|
|
28
|
+
return None
|
|
29
|
+
try:
|
|
30
|
+
value = json.loads(path.read_text(encoding="utf-8")).get("spent")
|
|
31
|
+
except (json.JSONDecodeError, AttributeError):
|
|
32
|
+
return None
|
|
33
|
+
return value if isinstance(value, (int, float)) and not isinstance(value, bool) else None
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _wait_for_pid(run_json: Path) -> dict:
|
|
37
|
+
"""Wait until `lab-kit run` has written this process's pid, so the two never write run.json at once."""
|
|
38
|
+
deadline = time.monotonic() + 10
|
|
39
|
+
while True:
|
|
40
|
+
data = records.read_json(run_json, str(run_json))
|
|
41
|
+
if data.get("pid") == os.getpid():
|
|
42
|
+
return data
|
|
43
|
+
if time.monotonic() > deadline:
|
|
44
|
+
data["pid"] = os.getpid()
|
|
45
|
+
records.write_json(run_json, data)
|
|
46
|
+
return data
|
|
47
|
+
time.sleep(0.02)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def supervise(run_dir: Path, lab_root: Path) -> int:
|
|
51
|
+
run_json = run_dir / records.RUN
|
|
52
|
+
data = _wait_for_pid(run_json)
|
|
53
|
+
env = records.script_env()
|
|
54
|
+
env[ENV_RUN] = str(run_dir)
|
|
55
|
+
env[ENV_OUT] = str(run_dir / "out")
|
|
56
|
+
env[ENV_WORK] = str(run_dir / "work")
|
|
57
|
+
with (run_dir / "run.log").open("a", encoding="utf-8") as log:
|
|
58
|
+
log.write(f"lab-kit: started {data['started']}: {' '.join(data['command'])}\n")
|
|
59
|
+
log.flush()
|
|
60
|
+
try:
|
|
61
|
+
code = subprocess.run(data["command"], cwd=lab_root, env=env, stdout=log, stderr=subprocess.STDOUT,
|
|
62
|
+
stdin=subprocess.DEVNULL, check=False).returncode
|
|
63
|
+
except OSError as exc:
|
|
64
|
+
log.write(f"lab-kit: the command could not start: {exc}\n")
|
|
65
|
+
code = 127
|
|
66
|
+
ended = records.now()
|
|
67
|
+
log.write(f"lab-kit: ended {ended} with exit code {code}\n")
|
|
68
|
+
run = records.RunRecord(data["protocol"], run_dir.name, run_dir, data)
|
|
69
|
+
records.write_manifest(run)
|
|
70
|
+
data = records.read_json(run_json, str(run_json))
|
|
71
|
+
data["ended"] = ended
|
|
72
|
+
data["exit"] = code
|
|
73
|
+
if data.get("spend"):
|
|
74
|
+
data["spent"] = _spent(run_dir / "out")
|
|
75
|
+
records.write_json(run_json, data)
|
|
76
|
+
return code
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
if __name__ == "__main__":
|
|
80
|
+
raise SystemExit(supervise(Path(sys.argv[1]), Path(sys.argv[2])))
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: lab-kit-cli
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A research lab's method and machinery on top of folio: pre-register, lock, run, score, re-derive.
|
|
5
|
+
Project-URL: Homepage, https://github.com/pierg/lab-kit
|
|
6
|
+
Project-URL: Issues, https://github.com/pierg/lab-kit/issues
|
|
7
|
+
Author: Piergiuseppe Mallozzi
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
|
+
Requires-Dist: folio-kb>=0.1.0
|
|
12
|
+
Requires-Dist: pyyaml
|
|
13
|
+
Provides-Extra: dev
|
|
14
|
+
Requires-Dist: pytest; extra == 'dev'
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
|
|
17
|
+
# lab-kit
|
|
18
|
+
|
|
19
|
+
lab-kit turns your coding agent into the staff of a research lab: it pre-registers each experiment, locks the plan before the first number, runs the lab's own code against the lock, scores the outcome against it, and records only numbers that re-derive from committed evidence. It is built on [folio](https://github.com/pierg/folio). folio decides what a document is; lab-kit decides when a result counts.
|
|
20
|
+
|
|
21
|
+
You do not run lab-kit yourself, and you do not walk the agent through each step. You install it by handing your agent one line, set up the lab with its question, and then plan each mission with the agent in plain words. Once you approve the plan, the agent runs it end to end: it drafts the protocol, has a reviewer read it, locks it, runs it, scores it, has the numbers re-derived by an independent reviewer, records the result and writes the report, running the gate before every commit. It stops only where the plan and the method say it must.
|
|
22
|
+
|
|
23
|
+
<picture>
|
|
24
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/pierg/lab-kit/main/docs/figures/how-lab-kit-works-dark.svg">
|
|
25
|
+
<img alt="How lab-kit works. You set up a lab for your question, plan a mission with your coding agent in plain words and approve it, then ask how it is going. The agent works through lab-kit's skills (set-up-lab, plan-mission, run-mission, experiment, review) and folio's skills for documents, and every change passes lab-kit check before it is committed. The lab in git has two zones: the folio library holds the question, the protocol, the result, the claim, the report and the journal; lab-kit's files hold the lock, the runs with sealed evidence, and the score. A protocol crosses into the lab files only through the lock, and a result comes back only through the reviewer pass." src="https://raw.githubusercontent.com/pierg/lab-kit/main/docs/figures/how-lab-kit-works-light.svg">
|
|
26
|
+
</picture>
|
|
27
|
+
|
|
28
|
+
## Install
|
|
29
|
+
|
|
30
|
+
Paste this into your coding agent, in the repository that will hold the lab:
|
|
31
|
+
|
|
32
|
+
```text
|
|
33
|
+
Install lab-kit here: run `uv tool install lab-kit-cli --with-executables-from folio-kb` (or `pipx install --include-deps lab-kit-cli`), then `lab-kit init`, then read .agents/skills/set-up-lab/SKILL.md and follow it.
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Installing lab-kit brings folio with it; the flag puts folio's command on the path too. The agent asks you at most three questions in one message (the lab's question, where the library lives, which paths are frozen), sets up the library with folio's lab pack, records the question as `Q-1`, and leaves the gate passing. A longer version of the prompt is in [SETUP.md](https://github.com/pierg/lab-kit/blob/main/SETUP.md). lab-kit needs Python 3.10 or later.
|
|
37
|
+
|
|
38
|
+
## Give it a mission
|
|
39
|
+
|
|
40
|
+
**1. Plan the mission together.** Once the lab is set up, state an objective:
|
|
41
|
+
|
|
42
|
+
```text
|
|
43
|
+
Mission: find out whether our pi estimator's error falls as 1/sqrt(n).
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
The **plan-mission** skill reads the lab, then asks in one message only what it cannot look up: "How many sample sizes and repeats count as enough? Token-free only? What would make you stop it early?" It drafts the plan: the question served, observable milestones, the scope, the spend, what it never touches, and where it must stop. You change what you want and say "approved". It records your approval and starts nothing.
|
|
47
|
+
|
|
48
|
+
**2. Let it run.** The **run-mission** skill carries out the approved plan alone; it refuses an unapproved one. It dispatches workers through the **experiment** and **review** skills, sends a reviewer before the protocol locks and before any result counts, and keeps `ops/STATE.md` current so a crashed session resumes from the record. It comes back to you only at the plan's stops, or when a run would spend past the cap, a change would touch a frozen surface, a lock or a recorded result, or the work would leave the plan's scope.
|
|
49
|
+
|
|
50
|
+
It reports the result by id, the misses as plainly as the hits, and its recommended next step. Meanwhile, ask "how is the mission going?" or "what should we do next?".
|
|
51
|
+
|
|
52
|
+
## The loop
|
|
53
|
+
|
|
54
|
+
<picture>
|
|
55
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/pierg/lab-kit/main/docs/figures/lab-loop-dark.svg">
|
|
56
|
+
<img alt="The lab loop: in the folio library, a question leads to a draft protocol; it enters lab-kit's files only through the lock, runs and is scored there, and comes back to the library only through the reviewer pass, where the result is re-derived; then the result, the claim and the report. The journal records the lock, the run, the result and the lesson or kill, and lab-kit check runs under everything." src="https://raw.githubusercontent.com/pierg/lab-kit/main/docs/figures/lab-loop-light.svg">
|
|
57
|
+
</picture>
|
|
58
|
+
|
|
59
|
+
The documents (questions, protocols, results, claims, reports, the journal) are folio's, through its lab pack. The steps between them are lab-kit's skills: **set-up-lab** (once), **plan-mission** (with you, until you approve), **run-mission** (alone: dispatch, supervise, recover, report), **experiment** (draft, lock, launch, watch) and **review** (fold, score, record, report). Four agent roles do the work: **scout**, **runner**, **reviewer** and **reporter**.
|
|
60
|
+
|
|
61
|
+
## The checks
|
|
62
|
+
|
|
63
|
+
The gate, `lab-kit check`, runs folio's checks and then the lab's. It runs offline, names every problem in one pass, and changes nothing. CI runs the same command.
|
|
64
|
+
|
|
65
|
+
- **Locks.** A locked protocol has a lock record, and its bytes still match it (`lab-lock-recorded`, `lab-lock-intact`).
|
|
66
|
+
- **Runs.** Every run started after its lock, used exactly the pinned configuration, and its evidence matches its manifest (`lab-run-after-lock`, `lab-roster-frozen`, `lab-evidence-sealed`).
|
|
67
|
+
- **Results.** Every live result re-derives exactly, rests on a finished run of a locked protocol, and passed an independent reviewer (`lab-rederive`, `lab-result-grounded`).
|
|
68
|
+
- **Scores.** A scorecard scores exactly the ids the lock names (`lab-score-exact`), and a scored experiment is reported or explained (`lab-scored-reported`).
|
|
69
|
+
- **Lab files.** Ids resolve, frozen surfaces are untouched, scripts pass `--selftest`, nothing runs under an unapproved mission, spend stays within its cap, records only grow, the method is current, and the state file stays a short pointer.
|
|
70
|
+
|
|
71
|
+
Together they hold a chain from every number on a page back to the lock:
|
|
72
|
+
|
|
73
|
+
<picture>
|
|
74
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/pierg/lab-kit/main/docs/figures/number-chain-dark.svg">
|
|
75
|
+
<img alt="The chain behind one number in the example lab: the report's slope of -0.493 cites result R-2; R-2's re-derive command must print -0.493 again from the committed estimates.tsv; the evidence still matches its manifest; the run's lock hash matches the lock record; and the run used exactly the protocol's pinned configuration. Each link names the check that holds it." src="https://raw.githubusercontent.com/pierg/lab-kit/main/docs/figures/number-chain-light.svg">
|
|
76
|
+
</picture>
|
|
77
|
+
|
|
78
|
+
The gate checks that the record is consistent. Whether it is honest is the reviewer's job.
|
|
79
|
+
|
|
80
|
+
## An example
|
|
81
|
+
|
|
82
|
+
[`examples/monte-carlo-lab`](https://github.com/pierg/lab-kit/blob/main/examples/monte-carlo-lab) asks one question: does the error of a Monte Carlo estimate of pi shrink as 1/sqrt(n)? Its protocol, locked before the run, draws 100 seeded estimates at each of five sample sizes from 64 to 16,384 points, in pure Python, in under a second. The RMS error fell with a fitted log-log slope of -0.493 (`R-2`), so the hypothesis is kept; two of the four predictions missed, and the report says so as plainly as it says the rest. `R-2` supersedes `R-1`, which fitted the wrong error measure. Every record in it was made by the loop above, and `lab-kit check` passes on it with no error and no warning.
|
|
83
|
+
|
|
84
|
+
Open it with your agent and ask "check this lab" or "re-derive R-2".
|
|
85
|
+
|
|
86
|
+
## Reference
|
|
87
|
+
|
|
88
|
+
The `lab-kit` command is the interface for agents and CI, the way git is: `init`, `check`, `experiment`, `lock`, `run`, `runs`, `score`, `rederive`, `freeze`, `status` and `version`. Each is specified in [`docs/spec/lab-model.md`](https://github.com/pierg/lab-kit/blob/main/docs/spec/lab-model.md) §6, the contract the skills, the roles, the checks and the command all build on. To work on lab-kit itself, see [CONTRIBUTING.md](https://github.com/pierg/lab-kit/blob/main/CONTRIBUTING.md). Changes: [CHANGELOG.md](https://github.com/pierg/lab-kit/blob/main/CHANGELOG.md).
|
|
89
|
+
|
|
90
|
+
## License
|
|
91
|
+
|
|
92
|
+
MIT. See [`LICENSE`](https://github.com/pierg/lab-kit/blob/main/LICENSE).
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
lab_kit/__init__.py,sha256=0FLgBfZ1lPGUCyQU-Q0VrOGG2A0QPXNP-Jmx20Ex9II,93
|
|
2
|
+
lab_kit/__main__.py,sha256=k1ocEWawweo1qCJWNFAAvyxz3tcY13dzvCenHszij30,48
|
|
3
|
+
lab_kit/cli.py,sha256=l91WGLFY3C-DHZQqBLi0U5R2OvnTytGQY2P1dVyruww,6006
|
|
4
|
+
lab_kit/data.py,sha256=Kh9PD5dWxVrbZC-PoVUxMVygAvsx5RiFC6XlLjlEx_4,1802
|
|
5
|
+
lab_kit/errors.py,sha256=gtLSSUhFAR-YDHEEw9PTO1itXMvfuDzzz1AE09fTBzM,115
|
|
6
|
+
lab_kit/frozen.py,sha256=_NOqDbFk1e8j7jauP0xIGxmVkITSgsgO5PrwIU_HJ5I,1758
|
|
7
|
+
lab_kit/gitlog.py,sha256=bZP4llSn4QoDOtkLTZVF_AHEdjLWI7Z9J8pUshkRZy0,2073
|
|
8
|
+
lab_kit/lab.py,sha256=8AMrjG3slUHnfTEclYc-pBFLTQZstMb5Z_0N-5V7Ypc,5073
|
|
9
|
+
lab_kit/ops.py,sha256=ZnDZqci3RqYaC0UOP9_71m22TryACiFhPJzgdiE0uJU,3077
|
|
10
|
+
lab_kit/protocol.py,sha256=oM351rcoRYdJSg-7GE7KC98czogdUT14KctoaWlvrng,3205
|
|
11
|
+
lab_kit/records.py,sha256=lF2QRbsmXKF_77F3okExEuXZs-oeyzDaPNamCFkuTMk,7576
|
|
12
|
+
lab_kit/rederive.py,sha256=seAXN7eYONkOLjGf9dX90sUKCW2aQ5P47GM855g1p6g,1903
|
|
13
|
+
lab_kit/supervise.py,sha256=nC9dLKD3KQPQyRXwwm5G9d2Jh8PLe5kzigsoorqbXN8,2820
|
|
14
|
+
lab_kit/checks/__init__.py,sha256=Hca4tFCyBEsroUmRcJ2lVnJ1F1feB8OBZqyrWmNTnCA,29
|
|
15
|
+
lab_kit/checks/base.py,sha256=oL1dGdYulOVUXFvEFZePY-0lfdG1VhLSDcCShXTECWs,1146
|
|
16
|
+
lab_kit/checks/files.py,sha256=QzgDfIIO33UfVZP2B6FJkG-GzKD7-fun9uUod7V3JyE,14767
|
|
17
|
+
lab_kit/checks/locks.py,sha256=II1AQ72MS3MFDpGQZ55M8o8puhCYeDbrTB7B--u-A3E,6597
|
|
18
|
+
lab_kit/checks/results.py,sha256=ygOYkSrA9qfJlN1ECnEls7aExAzENT1qL0CnZYgprkg,8095
|
|
19
|
+
lab_kit/checks/run.py,sha256=_SkjfXf4b2LeB_zE3ZJ4ckkgTAw2Y8RAWBMQQ_z78I4,2595
|
|
20
|
+
lab_kit/commands/__init__.py,sha256=hOovTB7uH2q2mSEA1mWVIUtQXAGqFJAYmT8Ms4wqNfk,74
|
|
21
|
+
lab_kit/commands/experiments.py,sha256=Mwd8MpwDw1vbeKWNw-jycltV4LhWI5iixpTkmQTYcNs,4032
|
|
22
|
+
lab_kit/commands/runs.py,sha256=jYHKnKZviAzfuqczq20C1_25kaZDevDPn5PvfXH0-_0,4611
|
|
23
|
+
lab_kit/commands/setup.py,sha256=BosaGQG7tRWDU8m7WekuNYNPNJ1FzSo9NX4bvJKsKFE,5113
|
|
24
|
+
lab_kit/commands/status.py,sha256=6bS8v9NbCwPv7SWWf7r4dgmoG1mo29HBQibUhwlKMFo,1183
|
|
25
|
+
lab_kit/_data/agents/reporter.md,sha256=Iv3SPcxLvkDI5LrISgWszdZcCakPwMJFPrPtT5X8YSA,2438
|
|
26
|
+
lab_kit/_data/agents/reviewer.md,sha256=URWaXS1YpfBmOBSUM1PQCRLzJqKQNdpx2fxmOUAbwo4,2697
|
|
27
|
+
lab_kit/_data/agents/runner.md,sha256=-V9HMMTWYNm-_KuR0_lzXkN0cWYA9vhnm7qRDQzl5b8,2248
|
|
28
|
+
lab_kit/_data/agents/scout.md,sha256=tdRsCLriqQJfeiBEMveCe2FctalRy1JjYJSj2Skvx-M,1512
|
|
29
|
+
lab_kit/_data/method/DISCIPLINE.md,sha256=lENc9Myq0d1FWbIYOLJ1Wp8YkCjQ61FTYAh3AFcqiQw,8287
|
|
30
|
+
lab_kit/_data/method/LADDER.md,sha256=F1YnykTdhrQw6URDxyN-7b9iPHd3ejgJOL8v9PZbR2M,5940
|
|
31
|
+
lab_kit/_data/skills/experiment/SKILL.md,sha256=BVBQRD8ATmQC9k46iUJOi1eNcuQ8TjXFxPymA1HDiSI,6290
|
|
32
|
+
lab_kit/_data/skills/plan-mission/SKILL.md,sha256=uTnfPQY_fwCMRJPXqhPtWKQhxMVFdmWJ56V_xyEdHBs,5782
|
|
33
|
+
lab_kit/_data/skills/review/SKILL.md,sha256=zRchLAOLx2fzhkZLWZt0czVvUzI5v8TV4FPld5k6A_U,6981
|
|
34
|
+
lab_kit/_data/skills/run-mission/SKILL.md,sha256=s_mMZsr3uhDzSatcSiP-jE-mFx-YjPGMeklmdqJK7h4,7785
|
|
35
|
+
lab_kit/_data/skills/set-up-lab/SKILL.md,sha256=SstrSgJoJs2aGdwoFa9P7kAFZh0EWMyJS3Xnj4oP_Cg,7902
|
|
36
|
+
lab_kit_cli-0.1.0.dist-info/METADATA,sha256=dpoZBGnnx5sXCSN_0QNmcF8zgaVB_HFQYYQ_wybXnrg,9029
|
|
37
|
+
lab_kit_cli-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
38
|
+
lab_kit_cli-0.1.0.dist-info/entry_points.txt,sha256=JYPwpDHfg-4r3J7zfcWjZolWcPiBkM0C1s34A4Lide4,45
|
|
39
|
+
lab_kit_cli-0.1.0.dist-info/licenses/LICENSE,sha256=R58jcDUMiqrgVefliI9qfiTqQU0-yTl1reIQSClNtug,1078
|
|
40
|
+
lab_kit_cli-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Piergiuseppe Mallozzi
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|