pi-crew 0.9.66 → 0.9.67

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/CHANGELOG.md +61 -0
  2. package/README.md +2 -3
  3. package/agents/executor.md +1 -1
  4. package/agents/test-engineer.md +1 -1
  5. package/agents/verifier.md +1 -1
  6. package/dist/index.mjs +4441 -4091
  7. package/package.json +7 -4
  8. package/scripts/analyze-run.mjs +1333 -0
  9. package/scripts/resource-sampler.mjs +482 -0
  10. package/skills/real-test-pi-crew/SKILL.md +10 -9
  11. package/src/config/role-tools.ts +2 -2
  12. package/src/extension/knowledge-injection.ts +17 -0
  13. package/src/extension/pi-api.ts +0 -16
  14. package/src/extension/team-tool/api/agent-control.ts +358 -0
  15. package/src/extension/team-tool/api/handler-context.ts +57 -0
  16. package/src/extension/team-tool/api/heartbeat.ts +75 -0
  17. package/src/extension/team-tool/api/mailbox.ts +242 -0
  18. package/src/extension/team-tool/api/plan-approval.ts +190 -0
  19. package/src/extension/team-tool/api/read.ts +443 -0
  20. package/src/extension/team-tool/api/task-claims.ts +207 -0
  21. package/src/extension/team-tool/api.ts +56 -1301
  22. package/src/extension/team-tool/cancel.ts +15 -4
  23. package/src/extension/team-tool/goal-wrap.ts +2 -2
  24. package/src/extension/team-tool/goal.ts +4 -4
  25. package/src/extension/team-tool/lifecycle-actions.ts +5 -5
  26. package/src/extension/team-tool/parallel-dispatch.ts +2 -2
  27. package/src/extension/team-tool/respond.ts +11 -3
  28. package/src/extension/team-tool/run-intent.ts +357 -0
  29. package/src/extension/team-tool/run.ts +8 -285
  30. package/src/extension/team-tool/status.ts +56 -21
  31. package/src/extension/team-tool.ts +10 -8
  32. package/src/prompt/scratchpad-lifecycle.ts +80 -5
  33. package/src/runtime/child-pi/child-pi-spawn.ts +13 -7
  34. package/src/runtime/child-pi/child-pi.ts +22 -144
  35. package/src/runtime/child-pi/mock-fixtures.ts +171 -0
  36. package/src/runtime/crew-agent-records.ts +18 -1
  37. package/src/runtime/scratchpad/README.md +6 -0
  38. package/src/runtime/scratchpad/guest.ts +54 -5
  39. package/src/runtime/scratchpad/snapshot-hmac.ts +7 -1
  40. package/src/runtime/supervisor-contact.ts +0 -16
  41. package/src/runtime/task-runner/child-executor.ts +1 -0
  42. package/src/runtime/task-runner/state-helpers.ts +9 -1
  43. package/src/runtime/team-runner.ts +39 -2
  44. package/src/state/contracts.ts +3 -0
  45. package/src/state/event-log/event-log.ts +19 -27
  46. package/src/state/gitignore-manager.ts +61 -7
  47. package/src/utils/glob-match.ts +29 -0
  48. package/types/dwf.d.ts +1 -1
  49. package/src/types/new-api-types.ts +0 -35
@@ -0,0 +1,482 @@
1
+ /**
2
+ * pi-crew resource sampler — external process monitor.
3
+ *
4
+ * Samples RSS / heap / CPU% for a PID and all its descendants, writing one
5
+ * JSONL line per PID per tick. Runs as a standalone external process so it
6
+ * works regardless of pi-crew's bundle/runtime (no src/ instrumentation).
7
+ *
8
+ * Modes:
9
+ * --watch-parent <pid> [--run-id <id>] [--interval 2000] [--out <path>]
10
+ * Polls <pid> + children until SIGINT/SIGTERM.
11
+ * --wrap <cmd...> [--run-id <id>] [--interval 2000] [--out <path>]
12
+ * Spawns <cmd>, samples it + children until exit, exits with child code.
13
+ *
14
+ * Output JSONL line: {ts,pid,ppid,label,rssBytes,heapBytes,cpuPct}
15
+ *
16
+ * Linux path uses /proc; non-Linux falls back to `ps -o rss=,pcpu=`.
17
+ *
18
+ * Run: node scripts/resource-sampler.mjs --wrap pi --version
19
+ */
20
+
21
+ import { spawn, spawnSync } from "node:child_process";
22
+ import { readFileSync, existsSync, mkdirSync, appendFileSync, readdirSync } from "node:fs";
23
+ import { join } from "node:path";
24
+
25
+ // ---------- CLI parsing ----------
26
+ // Known options may appear ANYWHERE (before or after --wrap). Everything
27
+ // after --wrap that is NOT a known option becomes the wrapped command.
28
+ const KNOWN_FLAGS = new Set(["--watch-parent", "--watch-run", "--crew-root", "--wrap", "--interval", "--run-id", "--out", "--no-live-warn", "-h", "--help"]);
29
+
30
+ function parseArgs(argv) {
31
+ const args = { interval: 2000, mode: null, parentPid: null, wrap: [], runId: null, out: null, watchRun: null, crewRoot: null, liveWarn: true };
32
+ let wrapStarted = false;
33
+ for (let i = 2; i < argv.length; i++) {
34
+ const a = argv[i];
35
+ // Once --wrap is seen, collect non-flag tokens as the command. A known
36
+ // flag is still parsed as an option (so `--wrap sleep 5 --interval 500`
37
+ // works), but an unknown token is treated as part of the command.
38
+ if (wrapStarted && !KNOWN_FLAGS.has(a)) {
39
+ args.wrap.push(a);
40
+ continue;
41
+ }
42
+ if (a === "--watch-parent") {
43
+ args.mode = "watch";
44
+ args.parentPid = Number.parseInt(argv[++i], 10);
45
+ } else if (a === "--watch-run") {
46
+ // auto-resolve the run's leader/runner PID from pi-crew state (no manual
47
+ // pgrep). Reads async.pid / manifest.async.pid / heartbeat.json.
48
+ args.mode = "watch";
49
+ args.watchRun = argv[++i];
50
+ if (!args.runId) args.runId = args.watchRun;
51
+ } else if (a === "--crew-root") {
52
+ args.crewRoot = argv[++i];
53
+ } else if (a === "--wrap") {
54
+ args.mode = "wrap";
55
+ wrapStarted = true;
56
+ } else if (a === "--interval") {
57
+ // R13 (audit): reject NaN and clamp tiny intervals — setInterval(fn, 0)
58
+ // or setInterval(fn, NaN) spins ~60-130×/s, writing huge files and
59
+ // burning CPU (each tick also scans all of /proc).
60
+ const raw = argv[++i];
61
+ const parsed = Number.parseInt(raw, 10);
62
+ if (Number.isNaN(parsed)) {
63
+ process.stderr.write(`Error: --interval must be a number (ms), got "${raw}"\n`);
64
+ process.exit(1);
65
+ }
66
+ if (parsed < 100) {
67
+ process.stderr.write(`[resource-sampler] --interval ${parsed}ms < 100ms; clamping to 100ms\n`);
68
+ args.interval = 100;
69
+ } else {
70
+ args.interval = parsed;
71
+ }
72
+ } else if (a === "--run-id") {
73
+ args.runId = argv[++i];
74
+ } else if (a === "--out") {
75
+ args.out = argv[++i];
76
+ } else if (a === "--no-live-warn") {
77
+ args.liveWarn = false;
78
+ } else if (a === "-h" || a === "--help") {
79
+ printHelp();
80
+ process.exit(0);
81
+ } else if (wrapStarted) {
82
+ // unknown token after --wrap → part of the command
83
+ args.wrap.push(a);
84
+ }
85
+ }
86
+ if (!args.mode) {
87
+ printHelp();
88
+ process.exit(1);
89
+ }
90
+ return args;
91
+ }
92
+
93
+ function printHelp() {
94
+ process.stderr.write(
95
+ [
96
+ "Usage:",
97
+ " resource-sampler.mjs --watch-parent <pid> [--run-id <id>] [--interval 2000] [--out <path>]",
98
+ " resource-sampler.mjs --wrap <cmd...> [--run-id <id>] [--interval 2000] [--out <path>]",
99
+ "",
100
+ ].join("\n"),
101
+ );
102
+ }
103
+
104
+ // ---------- /proc readers (Linux) ----------
105
+ const CLK_TCK = 100; // sysconf(_SC_CLK_TCK) on essentially all Linux/x86/arm
106
+
107
+ function readProcStat(pid) {
108
+ // /proc/<pid>/stat — comm (field 2) may contain spaces inside parens.
109
+ let raw;
110
+ try {
111
+ raw = readFileSync(`/proc/${pid}/stat`, "utf8");
112
+ } catch {
113
+ return null;
114
+ }
115
+ const open = raw.indexOf("(");
116
+ const close = raw.lastIndexOf(")");
117
+ if (open < 0 || close < 0) return null;
118
+ const comm = raw.slice(open + 1, close);
119
+ const rest = raw.slice(close + 2).trim().split(/\s+/);
120
+ // rest[0] = field3 (state), rest[1] = field4 (ppid), rest[11] = utime(f14), rest[12] = stime(f15), rest[19] = starttime(f22)
121
+ return {
122
+ pid,
123
+ comm,
124
+ state: rest[0],
125
+ ppid: Number.parseInt(rest[1], 10) || 0,
126
+ utime: Number.parseInt(rest[11], 10) || 0,
127
+ stime: Number.parseInt(rest[12], 10) || 0,
128
+ starttime: Number.parseInt(rest[19], 10) || 0,
129
+ };
130
+ }
131
+
132
+ function readProcStatus(pid) {
133
+ let raw;
134
+ try {
135
+ raw = readFileSync(`/proc/${pid}/status`, "utf8");
136
+ } catch {
137
+ return null;
138
+ }
139
+ let rssKb = 0;
140
+ let dataKb = 0;
141
+ for (const line of raw.split("\n")) {
142
+ if (line.startsWith("VmRSS:")) rssKb = Number.parseInt(line.slice(6).trim(), 10) || 0;
143
+ else if (line.startsWith("VmData:")) dataKb = Number.parseInt(line.slice(7).trim(), 10) || 0;
144
+ }
145
+ return { rssKb, dataKb };
146
+ }
147
+
148
+ // ---------- fallback via ps (non-Linux) ----------
149
+ function readPs(pid) {
150
+ const res = spawnSync("ps", ["-o", "rss=,pcpu=", "-p", String(pid)], { encoding: "utf8" });
151
+ if (res.status !== 0 || !res.stdout.trim()) return null;
152
+ const parts = res.stdout.trim().split(/\s+/);
153
+ return {
154
+ pid,
155
+ ppid: 0,
156
+ rssKb: Number.parseInt(parts[0], 10) || 0,
157
+ dataKb: 0,
158
+ cpuPct: Number.parseFloat(parts[1]) || 0,
159
+ ps: true,
160
+ };
161
+ }
162
+
163
+ // ---------- child discovery ----------
164
+ function findDescendants(rootPid) {
165
+ // BFS over /proc to find all PIDs whose ppid chain leads to rootPid.
166
+ if (!existsSync("/proc")) return [rootPid];
167
+ const all = [];
168
+ try {
169
+ for (const name of readdirSync("/proc")) {
170
+ if (/^\d+$/.test(name)) all.push(Number.parseInt(name, 10));
171
+ }
172
+ } catch {
173
+ return [rootPid];
174
+ }
175
+ const ppidOf = new Map();
176
+ for (const pid of all) {
177
+ const s = readProcStat(pid);
178
+ if (s) ppidOf.set(pid, s.ppid);
179
+ }
180
+ const result = new Set([rootPid]);
181
+ let frontier = [rootPid];
182
+ while (frontier.length) {
183
+ const next = [];
184
+ for (const [pid, ppid] of ppidOf) {
185
+ if (result.has(ppid) && !result.has(pid)) {
186
+ result.add(pid);
187
+ next.push(pid);
188
+ }
189
+ }
190
+ frontier = next;
191
+ }
192
+ return [...result];
193
+ }
194
+
195
+ // ---------- sampling ----------
196
+ const cpuPrev = new Map(); // pid -> {ticks, ts}
197
+
198
+ // ---- live anomaly warnings (stderr, rate-limited) ----
199
+ // Emitted DURING sampling so the user sees resource anomalies in real time,
200
+ // not only in the post-hoc analyze-run report. Rate-limited per (pid,category)
201
+ // so a sustained spike doesn't spam every tick.
202
+ const rssPrev = new Map(); // pid -> { rss, ts } (for one-interval RSS jump)
203
+ const rssHist = new Map(); // pid -> number[] (recent rssBytes, for slow-leak trend)
204
+ const warnCooldown = new Map(); // `${pid}:${cat}` -> last emit ts
205
+ const LIVE_CPU_PCT = 300; // single-process CPU% (multi-core: 300% = 3 cores)
206
+ const LIVE_RSS_JUMP = 200_000_000; // +200MB in one interval (fast spike)
207
+ const LIVE_RSS_HIGH = 1_000_000_000; // 1GB absolute
208
+ const LIVE_LEAK_WINDOW = 30; // samples to evaluate slow-leak trend (30s at 1s interval)
209
+ const LIVE_LEAK_GROWTH = 100_000_000; // +100MB net over the window = leak
210
+ const LIVE_LEAK_MONOTONIC = 0.75; // ≥75% of pairs increasing = monotonic-ish
211
+ const WARN_COOLDOWN_MS = 10_000;
212
+ let liveWarnEnabled = true;
213
+ function liveWarn(pid, label, cat, msg) {
214
+ if (!liveWarnEnabled) return;
215
+ const key = `${pid}:${cat}`;
216
+ const now = Date.now();
217
+ if (now - (warnCooldown.get(key) || 0) < WARN_COOLDOWN_MS) return;
218
+ warnCooldown.set(key, now);
219
+ process.stderr.write(`[resource-sampler] ⚠️ LIVE ${new Date(now).toISOString().slice(11, 19)} ${cat} pid=${pid}${label ? ` (${label})` : ""}: ${msg}\n`);
220
+ }
221
+
222
+ function samplePid(pid, rootPid) {
223
+ // Linux /proc path
224
+ if (existsSync(`/proc/${pid}/stat`)) {
225
+ const stat = readProcStat(pid);
226
+ const status = readProcStatus(pid);
227
+ if (!stat) return null;
228
+ const now = Date.now();
229
+ const ticks = stat.utime + stat.stime;
230
+ let cpuPct = 0;
231
+ const prev = cpuPrev.get(pid);
232
+ // firstSample: no usable baseline yet (first-ever sample OR PID reused).
233
+ // cpuPct is 0 here by construction — consumers should exclude it from CPU
234
+ // averages so short-lived subagents aren't dragged down by the first tick.
235
+ const firstSample = !prev || prev.starttime !== stat.starttime;
236
+ // PID-reuse guard: if starttime changed, a NEW process recycled this PID
237
+ // (common under respawn churn) — the old cpuPrev ticks are stale and would
238
+ // underreport. Treat as a first sample (cpuPct=0) and reset the baseline.
239
+ if (!firstSample) {
240
+ const dtick = ticks - prev.ticks;
241
+ const dsec = (now - prev.ts) / 1000;
242
+ if (dsec > 0) cpuPct = (dtick / CLK_TCK / dsec) * 100;
243
+ }
244
+ cpuPrev.set(pid, { ticks, ts: now, starttime: stat.starttime });
245
+ return {
246
+ ts: now,
247
+ pid,
248
+ ppid: stat.ppid,
249
+ label: pid === rootPid ? "root" : "child",
250
+ rssBytes: status ? status.rssKb * 1024 : 0,
251
+ heapBytes: status ? status.dataKb * 1024 : 0,
252
+ cpuPct: Math.max(0, Math.round(cpuPct * 10) / 10),
253
+ firstSample,
254
+ };
255
+ }
256
+ // fallback: ps
257
+ const ps = readPs(pid);
258
+ if (!ps) return null;
259
+ return {
260
+ ts: Date.now(),
261
+ pid,
262
+ ppid: ps.ppid,
263
+ label: pid === rootPid ? "root" : "child",
264
+ rssBytes: ps.rssKb * 1024,
265
+ heapBytes: 0,
266
+ cpuPct: Math.round(ps.cpuPct * 10) / 10,
267
+ };
268
+ }
269
+
270
+ function tick(rootPid, outPath) {
271
+ const pids = findDescendants(rootPid);
272
+ const seen = new Set(pids);
273
+ for (const pid of pids) {
274
+ const sample = samplePid(pid, rootPid);
275
+ if (sample) {
276
+ appendFileSync(outPath, JSON.stringify(sample) + "\n");
277
+ // live anomaly checks (rate-limited, stderr) — real-time observability
278
+ // zombie/dead child still in /proc (parent not reaping) — flag it live
279
+ if (pid !== rootPid && !isAlive(pid)) liveWarn(pid, sample.label, "proc_zombie", "process zombie/dead (state Z/X) — parent not reaping child");
280
+ if (sample.cpuPct >= LIVE_CPU_PCT) liveWarn(pid, sample.label, "high_cpu", `CPU ${sample.cpuPct}% (≥${LIVE_CPU_PCT}%)`);
281
+ const rp = rssPrev.get(pid);
282
+ if (rp) {
283
+ const jump = sample.rssBytes - rp.rss;
284
+ if (jump >= LIVE_RSS_JUMP) liveWarn(pid, sample.label, "rss_jump", `RSS +${(jump / 1024 / 1024).toFixed(0)}MB (${(rp.rss / 1024 / 1024).toFixed(0)}→${(sample.rssBytes / 1024 / 1024).toFixed(0)}MB) in one interval`);
285
+ }
286
+ rssPrev.set(pid, { rss: sample.rssBytes, ts: sample.ts });
287
+ if (sample.rssBytes >= LIVE_RSS_HIGH) liveWarn(pid, sample.label, "rss_high", `RSS ${(sample.rssBytes / 1024 / 1024).toFixed(0)}MB (≥1GB)`);
288
+ // slow-leak trend: gradual monotonic growth that no single-interval jump
289
+ // catches (e.g. +5MB/interval × 50). LONG window (30 samples) so warmup
290
+ // growth (V8 heap fill ~10-15s then plateau) does NOT fire — only
291
+ // SUSTAINED growth (30s+) is a leak signal. Validated on real e2e data:
292
+ // subagent warmup plateaus (no fire), long-lived session growth fires.
293
+ const h = rssHist.get(pid) || [];
294
+ h.push(sample.rssBytes);
295
+ if (h.length > LIVE_LEAK_WINDOW) h.shift();
296
+ rssHist.set(pid, h);
297
+ if (h.length >= LIVE_LEAK_WINDOW) {
298
+ const growth = h[h.length - 1] - h[0];
299
+ let inc = 0;
300
+ for (let i = 1; i < h.length; i++) if (h[i] > h[i - 1]) inc++;
301
+ if (growth >= LIVE_LEAK_GROWTH && inc / (h.length - 1) >= LIVE_LEAK_MONOTONIC) {
302
+ liveWarn(pid, sample.label, "rss_leak", `slow RSS leak +${(growth / 1024 / 1024).toFixed(0)}MB over ${h.length} samples (${(h[0] / 1024 / 1024).toFixed(0)}→${(h[h.length - 1] / 1024 / 1024).toFixed(0)}MB, monotonic) — possible memory leak`);
303
+ }
304
+ }
305
+ } else {
306
+ rssPrev.delete(pid);
307
+ rssHist.delete(pid);
308
+ }
309
+ }
310
+ // F4 (audit): prune cpuPrev entries for PIDs no longer alive — prevents the
311
+ // Map from accumulating stale entries for short-lived children over a long
312
+ // sampling session.
313
+ for (const pid of cpuPrev.keys()) {
314
+ if (!seen.has(pid)) {
315
+ cpuPrev.delete(pid);
316
+ rssPrev.delete(pid);
317
+ rssHist.delete(pid);
318
+ }
319
+ }
320
+ // prune warnCooldown for gone PIDs (keep rootPid entry)
321
+ for (const key of warnCooldown.keys()) {
322
+ const pid = Number(key.split(":")[0]);
323
+ if (!seen.has(pid) && pid !== rootPid) warnCooldown.delete(key);
324
+ }
325
+ }
326
+
327
+ // R4/sampler-test (audit): robust liveness — a process that exited but whose
328
+ // parent can't reap it yet is a ZOMBIE, and /proc/<pid>/stat still exists for
329
+ // zombies. Checking only file existence would never detect death in that
330
+ // case. Treat state 'Z' (zombie) and 'X' (dead) as not-alive.
331
+ function isAlive(pid) {
332
+ if (existsSync("/proc")) {
333
+ const stat = readProcStat(pid);
334
+ if (!stat) return false;
335
+ return stat.state !== "Z" && stat.state !== "X";
336
+ }
337
+ return readPs(pid) != null;
338
+ }
339
+ // Resolve a run's leader/runner PID from pi-crew state. Tries async.pid (a
340
+ // dedicated JSON file), manifest.async.pid, and heartbeat.json.pid. Polls
341
+ // briefly (the file may not exist the instant the run starts).
342
+ function resolveRunnerPid(runDir) {
343
+ const readJsonSafe = (p) => {
344
+ try {
345
+ return JSON.parse(readFileSync(p, "utf8"));
346
+ } catch {
347
+ return null;
348
+ }
349
+ };
350
+ for (let i = 0; i < 20; i++) {
351
+ const asyncPid = readJsonSafe(join(runDir, "async.pid"));
352
+ if (asyncPid && typeof asyncPid.pid === "number") return asyncPid.pid;
353
+ const manifest = readJsonSafe(join(runDir, "manifest.json"));
354
+ if (manifest && manifest.async && typeof manifest.async.pid === "number") return manifest.async.pid;
355
+ const hb = readJsonSafe(join(runDir, "heartbeat.json"));
356
+ if (hb && typeof hb.pid === "number") return hb.pid;
357
+ // R7 (audit): block-wait instead of spawning a throwaway node process
358
+ // every poll. Atomics.wait on the main thread is permitted in Node.
359
+ const waitBuf = new Int32Array(new SharedArrayBuffer(4));
360
+ Atomics.wait(waitBuf, 0, 0, 500);
361
+ }
362
+ return null;
363
+ }
364
+ function main() {
365
+ const args = parseArgs(process.argv);
366
+ liveWarnEnabled = args.liveWarn;
367
+ const outDir = join(process.cwd(), "bench", "results");
368
+ mkdirSync(outDir, { recursive: true });
369
+ const outFile =
370
+ args.out ||
371
+ join(outDir, `${args.runId || "wrap-" + Date.now()}.resources.jsonl`);
372
+ process.stderr.write(`[resource-sampler] writing → ${outFile}\n`);
373
+
374
+ if (args.mode === "wrap") {
375
+ if (args.wrap.length === 0) {
376
+ process.stderr.write("--wrap requires a command\n");
377
+ process.exit(1);
378
+ }
379
+ const child = spawn(args.wrap[0], args.wrap.slice(1), { stdio: "inherit" });
380
+ const rootPid = child.pid;
381
+ let exited = false;
382
+ // sample immediately, then on interval
383
+ tick(rootPid, outFile);
384
+ const handle = setInterval(() => tick(rootPid, outFile), args.interval);
385
+ const cleanup = () => {
386
+ clearInterval(handle);
387
+ // F3 (audit): kill the wrapped child so it is not orphaned when the
388
+ // sampler is signalled (SIGINT/SIGTERM). tryKill is best-effort.
389
+ try {
390
+ if (!exited) child.kill("SIGTERM");
391
+ } catch {
392
+ /* already gone */
393
+ }
394
+ };
395
+ process.on("SIGINT", () => {
396
+ cleanup();
397
+ process.exit(130);
398
+ });
399
+ process.on("SIGTERM", () => {
400
+ cleanup();
401
+ process.exit(143);
402
+ });
403
+ child.on("exit", (code, signal) => {
404
+ exited = true;
405
+ clearInterval(handle);
406
+ // final sample
407
+ tick(rootPid, outFile);
408
+ process.stderr.write(`[resource-sampler] child exited code=${code} signal=${signal}\n`);
409
+ process.exit(code ?? 1);
410
+ });
411
+ } else {
412
+ // watch-parent (or --watch-run, which resolves the runner PID from state)
413
+ let rootPid = args.parentPid;
414
+ if (args.watchRun) {
415
+ const crew = args.crewRoot || join(process.env.HOME || "/home/bom", ".crew");
416
+ const runDir = join(crew, "state", "runs", args.watchRun);
417
+ rootPid = resolveRunnerPid(runDir);
418
+ if (!rootPid) {
419
+ process.stderr.write(`--watch-run: could not resolve runner PID for ${args.watchRun} in ${runDir} (async.pid / manifest.async.pid / heartbeat.json)\n`);
420
+ process.exit(1);
421
+ }
422
+ process.stderr.write(`[resource-sampler] --watch-run ${args.watchRun} → runner PID ${rootPid}\n`);
423
+ }
424
+ if (!rootPid) {
425
+ process.stderr.write("--watch-parent requires a PID\n");
426
+ process.exit(1);
427
+ }
428
+ // R4 (audit): auto-stop when the watched tree dies. Previously the sampler
429
+ // kept ticking forever (writing nothing) after rootPid exited, until the
430
+ // user remembered to Ctrl-C. Now: if rootPid is not alive for 3
431
+ // consecutive ticks, stop cleanly. rootPid (the team leader) outlives the
432
+ // whole run, so its death = run over.
433
+ // PLUS (perf-obs): in --watch-run mode the runner PID may be the
434
+ // long-lived foreground pi process (sync runs) — it stays alive after the
435
+ // run, so ALSO stop when the run's manifest.json reaches a terminal
436
+ // status (completed/failed/cancelled/blocked).
437
+ const runDir = args.watchRun ? join(args.crewRoot || join(process.env.HOME || "/home/bom", ".crew"), "state", "runs", args.watchRun) : null;
438
+ const runIsTerminal = () => {
439
+ if (!runDir) return false;
440
+ try {
441
+ const m = JSON.parse(readFileSync(join(runDir, "manifest.json"), "utf8"));
442
+ return ["completed", "failed", "cancelled", "blocked"].includes(m.status);
443
+ } catch {
444
+ return false;
445
+ }
446
+ };
447
+ let deadTicks = 0;
448
+ const tickWatch = () => {
449
+ const alive = isAlive(rootPid);
450
+ if (alive) {
451
+ deadTicks = 0;
452
+ if (runIsTerminal()) {
453
+ if (handle) clearInterval(handle);
454
+ process.stderr.write(`[resource-sampler] run ${args.watchRun} reached terminal status — stopping\n`);
455
+ process.exit(0);
456
+ }
457
+ tick(rootPid, outFile);
458
+ } else {
459
+ deadTicks++;
460
+ if (deadTicks === 1) liveWarn(rootPid, "root", "proc_died", `watched PID ${rootPid} no longer alive`);
461
+ if (deadTicks >= 3) {
462
+ clearInterval(handle);
463
+ process.stderr.write(`[resource-sampler] watched PID ${rootPid} gone — stopping\n`);
464
+ process.exit(0);
465
+ }
466
+ }
467
+ };
468
+ let handle;
469
+ tickWatch();
470
+ handle = setInterval(tickWatch, args.interval);
471
+ const shutdown = () => {
472
+ clearInterval(handle);
473
+ tick(rootPid, outFile);
474
+ process.stderr.write("[resource-sampler] stopped\n");
475
+ process.exit(0);
476
+ };
477
+ process.on("SIGINT", shutdown);
478
+ process.on("SIGTERM", shutdown);
479
+ }
480
+ }
481
+
482
+ main();
@@ -110,7 +110,7 @@ To add Tier 1 to CI as a fast-feedback gate (under 30s):
110
110
 
111
111
  ---
112
112
 
113
- ## Tier 1 — Critical unit tests (~21s, 97 tests, the only suite you need for broker/UI changes)
113
+ ## Tier 1 — Critical unit tests (~25s, 101 tests, the only suite you need for broker/UI changes)
114
114
 
115
115
  **What**: run the curated 14-file fast subset.
116
116
 
@@ -122,7 +122,7 @@ To add Tier 1 to CI as a fast-feedback gate (under 30s):
122
122
  time npm run test:critical
123
123
  ```
124
124
 
125
- Expected output: `# tests 97 # pass 97 # fail 0 # duration_ms ~21000`.
125
+ Expected output: `# tests 101 # pass 101 # fail 0 # duration_ms ~26000`. (Count was 97 at v0.9.46; 101 since the model-routing merge — verify with the actual run; the skill's hard-coded numbers drift between releases.)
126
126
 
127
127
  **References**:
128
128
 
@@ -162,7 +162,7 @@ PI_CREW_BROKER=0 npm run test:critical
162
162
  PI_CREW_BROKER=1 npm run test:critical
163
163
  ```
164
164
 
165
- All three must show `# pass 97 # fail 0`. Measured times in this session: ~20s for default and `PI_CREW_BROKER=0`, ~21s for `PI_CREW_BROKER=1` (varies ±1s run-to-run).
165
+ All three must show `# pass 101 # fail 0`. Measured times in this session (2026-08-11): ~26s for default, ~26s for `PI_CREW_BROKER=0`, ~26s for `PI_CREW_BROKER=1` (varies ±1-2s run-to-run).
166
166
 
167
167
  **References**:
168
168
 
@@ -202,7 +202,7 @@ Compare the printed md5 against what the user's Pi session loaded. If they diffe
202
202
  | Bundle builder | `scripts/build-bundle.mjs` (esbuild-based, bundles `index.bundle.ts` → `dist/index.mjs`) |
203
203
  | Bundle resolution rule | `index.ts:5-22` (entrypoint docstring); also `scripts/build-bundle.mjs:14-20` (entrypoint preference); **symlink is live for source files but the bundled `dist/index.mjs` is loaded** |
204
204
  | Postinstall hook | `scripts/postinstall.mjs:43` — best-effort bundle rebuild; falls back to strip-types if esbuild missing |
205
- | Bundle md5 after Phase-4 commit | `1cc4d55e18add7b9a036c569143320b6` (~2.78 MB at the time; bundle size drifts ±5% between releases, check current `ls -la dist/index.mjs`) |
205
+ | Bundle md5 after Phase-4 commit | `1cc4d55e18add7b9a036c569143320b6` (~2.78 MB at the time; **check current**: `md5sum dist/index.mjs`. As of v0.9.66 I-batch 2026-08-11: `16e29d053bd370e24f40df147dadcb79` ~2.81 MB) |
206
206
 
207
207
  ---
208
208
 
@@ -454,7 +454,7 @@ If the two md5s match → session is on the latest code. If not → user must `/
454
454
 
455
455
  **Real measured outcome** (this session, after the v0.9.57 schema fix): 9a (15 team actions) + 9b (4 subagent tools / 3 run paths) exercised; all green; the two silent-failure modes that motivated this tier (`Unknown type` from `Type.Unsafe` without Kind, and `Validation failed for tool team` from empty-string-strict schema) were caught ONLY by this battery — Tier 1-8 all passed while the team tool was broken live. The session also surfaced the unauthorized-agent-edit anti-pattern (a chain-run agent edited `chain-runner.ts` mid-smoke) — see Anti-patterns.
456
456
 
457
- **Not covered by the cheap battery above** — the actions below need extra setup, cost, or user confirmation. Run them only when the change touches their code path, and prefer a throwaway cwd / config so you don't mutate the user's real state. Organised by cost/safety:
457
+ **Not covered by the cheap battery above** — the actions below need extra setup, cost, or user confirmation. Run them only when the change touches their code path, and prefer a throwaway cwd / config so you don't mutate the user's real state. Organised by cost/safety. **As of 2026-08-11 (extended battery, run report `real-test-2026-08-11-scratchpad-I-batch.md`), 9c/9e/9f have been exercised live once each — they are no longer unproven, but still require explicit scope+confirmation to re-run.**
458
458
 
459
459
  **9c. Lifecycle / recovery** (needs a *running* run — start an async run, then exercise these against its runId):
460
460
  - `team action='wait' runId='...'` — block until completion
@@ -508,9 +508,10 @@ If the two md5s match → session is on the latest code. If not → user must `/
508
508
  | `makeFakeCtx({ flagOn: false })` without `brokerEnv: "0"` | `makeFakeCtx` deletes `PI_CREW_BROKER` env if `brokerEnv` is undefined | `612e18b` (test fix) | `test/unit/crew-broker-server-gate.test.ts:78` — pass `brokerEnv: "0"` to preserve env |
509
509
  | Trust green CI on one OS | macOS/Windows regressions slip through | n/a (permanent) | `.crew/knowledge.md` — "CI runs 3 OSes ... A flake on one OS IS a real bug" |
510
510
  | Trusting a team-run agent not to edit the repo under test | Agents spawned by `team`/`Agent`/`crew_agent` inherit the session cwd and have `edit`/`write` tools — a proactive LLM (observed with deepseek) will make **unauthorized source edits** to pi-crew during a trivial smoke run (e.g. "improving" `chain-runner.ts` while parsing a chain string). The edit can be correct + green-tested yet still be unintended scope creep that silently lands in your commit. | n/a (permanent) | After EVERY team/subagent run: `git status` and verify each changed file was authored by you. Diff + review any surprise change before staging. Consider `workspaceMode: 'worktree'` for parallel/risky runs to isolate mutations. |
511
+ | **Armed-role tool-surface bug (found live 2026-08-11)**: an opt-in tool (e.g. `scratchpad`) is armed via `ROLE_TOOL_CONFIGS[role].scratchpad=true` AND env `PI_CREW_SCRATCHPAD=1`, but NEVER appears in the worker surface. Root cause: `resolveToolPolicy` (`src/agents/agent-config.ts:165`) falls back `roleConfig.tools ?? agent.tools` when the role has no `tools` allowlist, and the builtin `agents/{executor,verifier,test-engineer}.md` frontmatter `tools:` did not list `scratchpad` → pi got `--tools read,grep,find,ls,bash,edit,write` and **hard-filtered** scratchpad. Env was correct; the tool was silently dropped by the `--tools` allowlist. Reproduce: `pi -p --tools read,grep,find,ls,bash,edit,write "list tools"` → no scratchpad; with scratchpad added → present. | `f753be30` | **Fix**: keep armed-role `agents/*.md` frontmatter `tools:` lists in sync with `ROLE_TOOL_CONFIGS` (QW17 pins the pinned roles; add the new tool to BOTH frontmatter AND role config for pinned roles, frontmatter-only for vacuous roles). A smoke run that claims the tool is "not available" in the worker is a REAL signal — verify the worker's actual `--tools` allowlist, not just env vars. |
511
512
  | `Type.Unsafe({ anyOf/type })` schema field **without** `[TypeBox.Kind]` symbol | `Value.Check` throws `Unknown type` the first time a model emits that field (e.g. `skill`, `config`) — every team action returns `isError:true` text `"Unknown type"`. Tier 1-8 stay green because unit tests never send the offending field. | v0.9.57 | `src/schema/team-tool-schema.ts` — `SkillOverride`/`FreeformConfig` switched from `Type.Unsafe` to TypeBox-native `Type.Union`/`Type.Record`. See Tier 9. |
512
513
  | Schema too strict for model-emitted empty strings (`runId:""`, `workspaceMode:""`, `budgetTotal:0`) | pi-ai `validateToolArguments` runs BEFORE the pi-crew handler and rejects `""` against Literal unions / patterns → `Validation failed for tool team` → model loops. | v0.9.57 | `src/schema/team-tool-schema.ts` — added `Literal("")` to unions, `^$|` pattern for runId, `""` to action enum, `0`/Boolean allowances. Handler-side `normalizeTeamParams` drops the empties. |
513
- | Claiming "all 9 tiers pass" while 9c–9f were never run | Overclaim — once reported "9 tiers pass" when only 9a (8/10) + 9b (4/5) had actually run; 9c–9f were skipped. Past runs then become unverifiable ("did it really pass 9 tiers?"). | n/a (process) | Fill `REPORT-TEMPLATE.md` per-tier DURING the run. "Tier 9 pass" = 9a AND 9b AND the applicable 9c–9f, each with evidence. Round-up-to-pass is the anti-pattern this row exists to prevent. |
514
+ | Claiming "all 9 tiers pass" while 9c–9f were never run | Overclaim — once reported "9 tiers pass" when only 9a (8/10) + 9b (4/5) had actually run; 9c–9f were skipped. Past runs then become unverifiable ("did it really pass 9 tiers?"). **2026-08-11 repeat**: an initial report said "9c–9f skipped" yet the summary read as full coverage until the gap was called out. | n/a (process) | Fill `REPORT-TEMPLATE.md` per-tier DURING the run. "Tier 9 pass" = 9a AND 9b AND the applicable 9c–9f, each with evidence. Round-up-to-pass is the anti-pattern this row exists to prevent. If 9c–9f are skipped, SAY SO in the verdict and do not phrase it as "all pass". |
514
515
  | chain run with `workflow:"chain"` forwarded to steps | Every chain step fails in ~58ms with an EMPTY error string — looks like a parse failure but isn't. `chain-dispatch` forwards `params.workflow` ("chain") into executor overrides; each step then runs the "chain" workflow via the normal `executeTeamRun` path and fails fast + silently. | Open (issue #44) | Omit `workflow` when invoking `action:'run' chain=...` — chain then runs 2/2 success (~308s). See `docs/bugs/chain-workflow-forward-quirk.md`. |
515
516
 
516
517
  ---
@@ -648,7 +649,7 @@ The skill does NOT need to be updated for every commit — only when the cited l
648
649
  ## Quick reference — exact commands
649
650
 
650
651
  ```bash
651
- # Tier 1 (critical unit, ~21s, 97 tests)
652
+ # Tier 1 (critical unit, ~25s, 101 tests)
652
653
  npm run test:critical
653
654
  # Tier 2 (3-path proof, broker changes only)
654
655
  PI_CREW_BROKER=0 npm run test:critical
@@ -687,14 +688,14 @@ md5sum "$(npm root -g)"/pi-crew/dist/index.mjs 2>/dev/null \
687
688
 
688
689
  Before claiming "tested":
689
690
 
690
- - [ ] Tier 1: `test:critical` fresh-run, all pass (<25s). Count varies by release — was 97 at v0.9.46, 101 after the model-routing merge; record the actual count in the report.
691
+ - [ ] Tier 1: `test:critical` fresh-run, all pass (<25s). Count varies by release — was 97 at v0.9.46, **101 since the model-routing merge (v0.9.66)**; record the actual count in the report.
691
692
  - [ ] Tier 2: 3-path proof all pass — **required if you touched `src/config/defaults.ts` or `src/extension/registration/lifecycle-handlers.ts`**
692
693
  - [ ] Tier 3: `npm run typecheck` exit 0, `npm run build:bundle` exit 0
693
694
  - [ ] Tier 4: bundle md5 matches what the session loaded (or user has `/quit`-ed + reopened)
694
695
  - [ ] Tier 5/6: live TUI smoke for any `src/ui/` change — keystroke reached `handleInput`
695
696
  - [ ] Tier 7: smoke team run for any `src/runtime/plan-templates.ts` or `workflows/*.workflow.md` change — completed, no hang, verifier output under 60s
696
697
  - [ ] Tier 8: final md5 sync check passed
697
- - [ ] Tier 9: feature battery — **required if you touched `src/schema/team-tool-schema.ts`, `src/extension/registration/team-tool.ts`, or any `Type.Unsafe({...})` schema**. 9a read-only batch all return clean; one probe per 9b spawn path (sync / async / chain / `Agent` / `crew_agent`+`get_subagent_result`) completes with `consistency=1`. Run 9c–9f only when the change touches their code path; 9d (destructive) requires explicit user confirmation. **After every run: `git status` to catch unauthorized agent edits.**
698
+ - [ ] Tier 9: feature battery — **required if you touched `src/schema/team-tool-schema.ts`, `src/extension/registration/team-tool.ts`, any `Type.Unsafe({...})` schema, or any armed-role tool list (`agents/*.md` / `src/config/role-tools.ts`)**. 9a read-only batch all return clean; one probe per 9b spawn path (sync / async / chain / `Agent` / `crew_agent`+`get_subagent_result`) completes with `consistency=1`. Run 9c–9f only when the change touches their code path; **at least one full 9c/9e/9f sweep per release is recommended so the battery stays proven** (see `real-test-2026-08-11-scratchpad-I-batch.md`); 9d (destructive) requires explicit user confirmation. **After every run: `git status` to catch unauthorized agent edits.**
698
699
  - [ ] **Output report**: save `docs/real-test/reports/real-test-<YYYY-MM-DD>-<slug>.md` from `skills/real-test-pi-crew/REPORT-TEMPLATE.md`, filled DURING the run with per-tier evidence (counts/md5/runId) — not reconstructed from memory afterward. This is what makes past runs verifiable instead of trust-the-summary.
699
700
 
700
701
  **"All 9 tiers pass" is a claim that needs per-row evidence.** Tier 9 means 9a **and** 9b **and** whichever of 9c–9f applies to the change — not "9a passed, therefore 9 passed". If any required item above is unchecked or lacks concrete evidence (a number, an md5, a runId), the answer to "is it tested?" is **no** — say so explicitly instead of rounding up to "pass".
@@ -84,7 +84,7 @@ export const ROLE_TOOL_CONFIGS: Record<string, RoleToolConfig> = {
84
84
  // agents/verifier.md). Tool-set keeps bash but excludes edit/write so source
85
85
  // integrity is preserved during verification. Mirrors cold-verifier behavior.
86
86
  verifier: {
87
- tools: ["read", "grep", "find", "ls", "bash"],
87
+ tools: ["read", "grep", "find", "ls", "bash", "scratchpad"],
88
88
  excludeTools: ["edit", "write", "web"],
89
89
  // Phase 1 scratchpad: multi-cell test/verify flows reuse parsed state.
90
90
  scratchpad: true,
@@ -92,7 +92,7 @@ export const ROLE_TOOL_CONFIGS: Record<string, RoleToolConfig> = {
92
92
 
93
93
  // Test Engineer - Can write tests (F1: hyphenated key)
94
94
  "test-engineer": {
95
- tools: ["read", "edit", "write", "bash", "ls"],
95
+ tools: ["read", "edit", "write", "bash", "ls", "scratchpad"],
96
96
  excludeTools: ["web"],
97
97
  // Phase 1 scratchpad: build/run test suites with state across cells.
98
98
  scratchpad: true,
@@ -301,6 +301,7 @@ export function readKnowledge(cwd: string, query?: KnowledgeQuery): string {
301
301
  content = `${content.slice(0, MAX_KNOWLEDGE_HEAD_BYTES)}\n\n<!-- knowledge.md truncated at ${MAX_KNOWLEDGE_HEAD_BYTES} bytes (head shown). Full file: ${p} — use the \`read\` tool if you need sections beyond the head. -->`;
302
302
  }
303
303
  knowledgeCache.set(p, { key: cacheKey, content });
304
+ enforceKnowledgeCacheCap(knowledgeCache);
304
305
  return content;
305
306
  }
306
307
 
@@ -325,6 +326,7 @@ export function readKnowledge(cwd: string, query?: KnowledgeQuery): string {
325
326
  conventions: parsed.conventions,
326
327
  sessionLog: parsed.sessionLog,
327
328
  });
329
+ enforceKnowledgeCacheCap(sectionCache);
328
330
  }
329
331
 
330
332
  const queryText = [query.goal, query.taskText].filter(Boolean).join(" \n ");
@@ -382,6 +384,21 @@ interface CachedSections {
382
384
  }
383
385
  const sectionCache = new Map<string, CachedSections>();
384
386
 
387
+ // H6 (2026-08-10): FIFO cap — mirrors the discover-* caches
388
+ // (TEAM_DISCOVERY_MAX_ENTRIES = 32, WORKFLOW_DISCOVERY_MAX_ENTRIES = 32).
389
+ // Without this, the caches grow with every distinct project root seen by the
390
+ // process; each entry can be KB-scale (full knowledge.md content + parsed
391
+ // sections). Eviction is safe: a cache miss falls through to a fresh
392
+ // readFileSync + parse, same as the first call for that path.
393
+ const KNOWLEDGE_CACHE_MAX_ENTRIES = 64;
394
+
395
+ function enforceKnowledgeCacheCap(cache: Map<string, unknown>): void {
396
+ if (cache.size > KNOWLEDGE_CACHE_MAX_ENTRIES) {
397
+ const oldest = cache.keys().next().value;
398
+ if (oldest !== undefined) cache.delete(oldest);
399
+ }
400
+ }
401
+
385
402
  /** Head cap for the no-query (legacy / main-session) path. */
386
403
  const MAX_KNOWLEDGE_HEAD_BYTES = 2_000;
387
404
 
@@ -38,19 +38,3 @@ export type {
38
38
  } from "@earendil-works/pi-coding-agent";
39
39
 
40
40
  export { createBashTool, defineTool } from "@earendil-works/pi-coding-agent";
41
-
42
- /**
43
- * @deprecated Drift detector removed in Phase 5 follow-up to H1.
44
- *
45
- * History: this constant was meant to surface version-drift between
46
- * dev-time type-check and runtime peer-dep version. pi-crew declares
47
- * peer deps as `*` so the runtime version is set by whatever host pi
48
- * install the user has — pinning this constant to a specific devDep
49
- * range was a false invariant that did not reflect how extensions
50
- * actually consume peer packages.
51
- *
52
- * Kept exported (now `"0.0.0-unset"`) only for any downstream consumers
53
- * that may import it via re-exports. No diagnostic uses it anymore.
54
- * Will be removed in a future major version.
55
- */
56
- export const BUILT_AGAINST_PI_VERSION = "0.0.0-unset";