@rallycry/conveyor-skills 1.0.12 → 1.0.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -80,6 +80,33 @@ and safe:
80
80
  - `conveyor-skills link --check` verifies the links without changing anything
81
81
  (exit 1 when out of date) — useful in CI.
82
82
 
83
+ ## Taking turns on one checkout
84
+
85
+ Local agent sessions that share one git checkout take turns on it.
86
+ `/conveyor-build`, `/conveyor-local-loop` and local `/conveyor-review` fixes
87
+ hold a **checkout claim** while they use the working tree; a second session
88
+ that finds it held says so and waits instead of switching branches underneath
89
+ the first. Cloud pods have their own checkout and skip this.
90
+
91
+ The claim is one file in the git dir (`.git/conveyor-checkout-claim.json`, or
92
+ the per-worktree git dir), so it is never committed and never shows in
93
+ `git status`. The skills drive it through the same bin:
94
+
95
+ ```bash
96
+ conveyor-skills checkout acquire --card <slug> [--branch <b>] # take / refresh
97
+ conveyor-skills checkout verify --card <slug> # still mine? (heartbeat)
98
+ conveyor-skills checkout release --card <slug> # give it back
99
+ conveyor-skills checkout status # who holds it
100
+ conveyor-skills checkout wait --timeout 1740 # block until free
101
+ ```
102
+
103
+ Each command prints one JSON line. A claim goes stale when its holder's
104
+ process is dead, when the same process started a new session, or when its
105
+ heartbeat is older than two hours (`CONVEYOR_CHECKOUT_CLAIM_TTL_SECONDS`).
106
+ `conveyor-skills checkout release --force` is the human override for a wedged
107
+ claim. The full protocol lives in
108
+ [`skills/conveyor-build/references/checkout-claim.md`](skills/conveyor-build/references/checkout-claim.md).
109
+
83
110
  ## Versioning
84
111
 
85
112
  Versions are cut automatically from git tags on the Conveyor repo's `main`
@@ -0,0 +1,524 @@
1
+ // conveyor-skills checkout — make local agent sessions that share one git
2
+ // checkout take turns on it. A session holds the claim while it uses the
3
+ // working tree; any other session that finds the claim held waits instead of
4
+ // switching branches underneath the holder.
5
+ //
6
+ // The claim is one JSON file in the per-worktree git dir
7
+ // (`git rev-parse --absolute-git-dir`), so it is never committed, never shows
8
+ // in `git status`, survives branch switches, and linked worktrees get
9
+ // independent claims. Pods have their own checkout and never call this.
10
+ //
11
+ // Every command prints exactly ONE JSON line on stdout; hints go to stderr.
12
+ // Node built-ins only — this ships in the published package next to cli.mjs.
13
+
14
+ import { spawnSync } from "node:child_process";
15
+ import {
16
+ mkdirSync,
17
+ readFileSync,
18
+ renameSync,
19
+ rmSync,
20
+ statSync,
21
+ unlinkSync,
22
+ writeFileSync,
23
+ } from "node:fs";
24
+ import { hostname as osHostname } from "node:os";
25
+ import { join } from "node:path";
26
+ import process from "node:process";
27
+
28
+ export const CLAIM_FILE = "conveyor-checkout-claim.json";
29
+ export const MUTEX_DIR = "conveyor-checkout-claim.mutex";
30
+ export const SCHEMA = 1;
31
+ export const DEFAULT_TTL_SECONDS = 7200;
32
+ export const DEFAULT_WAIT_TIMEOUT_SECONDS = 1740;
33
+ export const DEFAULT_WAIT_INTERVAL_SECONDS = 5;
34
+ const MUTEX_RETRY_MS = 2000;
35
+ const MUTEX_STALE_MS = 30_000;
36
+
37
+ export const EXIT = { ok: 0, error: 1, held: 3, branchMoved: 4, dirty: 5 };
38
+
39
+ const USAGE = `Usage: conveyor-skills checkout <command> [options]
40
+
41
+ Commands:
42
+ acquire --card <slug> [--branch <b>] take or refresh the checkout claim
43
+ verify --card <slug> [--branch <b>] confirm the claim is still yours (also a heartbeat)
44
+ release --card <slug> [--force] drop the claim (--force: drop someone else's)
45
+ status show the claim; always exits 0
46
+ wait [--timeout 1740] [--interval 5] block until the claim is free, stale, or yours
47
+
48
+ Options:
49
+ --session <id> session identity (default: CONVEYOR_SESSION_ID, then
50
+ CLAUDE_CODE_SESSION_ID, then card:<slug>)
51
+ --ttl <secs> heartbeat age after which a claim is stale
52
+ (default: CONVEYOR_CHECKOUT_CLAIM_TTL_SECONDS or ${DEFAULT_TTL_SECONDS})
53
+
54
+ Exit codes: 0 ok, 1 usage error or not a git repo, 3 held by another session
55
+ (or lost), 4 branch moved under the claim, 5 dirty tree.
56
+ `;
57
+
58
+ /** Marker for a claim file that exists but does not parse as a claim. */
59
+ export const CORRUPT = Object.freeze({ corrupt: true });
60
+
61
+ /**
62
+ * Who is asking. The session id is what owns a claim; the pid (the Claude Code
63
+ * process, when the harness exports it) is what lets a same-host caller prove
64
+ * a holder is dead instead of waiting out the TTL.
65
+ */
66
+ export function resolveIdentity({ session, card, env = process.env } = {}) {
67
+ const resolved =
68
+ session ||
69
+ env.CONVEYOR_SESSION_ID ||
70
+ env.CLAUDE_CODE_SESSION_ID ||
71
+ (card ? `card:${card}` : null);
72
+ const pid = Number.parseInt(env.CLAUDE_PID ?? "", 10);
73
+ return { session: resolved, pid: Number.isInteger(pid) && pid > 0 ? pid : null };
74
+ }
75
+
76
+ export function defaultPidAlive(pid) {
77
+ try {
78
+ process.kill(pid, 0);
79
+ return true;
80
+ } catch (err) {
81
+ // EPERM: the process exists but belongs to someone else — alive.
82
+ return err?.code === "EPERM";
83
+ }
84
+ }
85
+
86
+ function isClaimRecord(value) {
87
+ return (
88
+ value !== null &&
89
+ typeof value === "object" &&
90
+ typeof value.session === "string" &&
91
+ typeof value.host === "string" &&
92
+ !Number.isNaN(Date.parse(value.heartbeatAt))
93
+ );
94
+ }
95
+
96
+ /**
97
+ * Classify an existing claim relative to the caller.
98
+ * kind: none | own | live | stale | corrupt. `reason` explains a stale verdict.
99
+ */
100
+ export function classifyClaim(claim, { identity, now, hostname, pidAlive, ttlSeconds }) {
101
+ if (claim === null || claim === undefined) return { kind: "none" };
102
+ if (!isClaimRecord(claim)) return { kind: "corrupt" };
103
+ const ageSeconds = Math.max(0, Math.round((now - Date.parse(claim.heartbeatAt)) / 1000));
104
+ if (identity.session && claim.session === identity.session) return { kind: "own", ageSeconds };
105
+ if (claim.host === hostname && Number.isInteger(claim.pid)) {
106
+ if (!pidAlive(claim.pid)) return { kind: "stale", reason: "pid-dead", ageSeconds };
107
+ // Same live process, new session id: the holder ran /clear and moved on.
108
+ if (identity.pid === claim.pid)
109
+ return { kind: "stale", reason: "same-pid-new-session", ageSeconds };
110
+ }
111
+ if (ageSeconds > ttlSeconds) return { kind: "stale", reason: "ttl-expired", ageSeconds };
112
+ return { kind: "live", ageSeconds };
113
+ }
114
+
115
+ /**
116
+ * What `acquire` does with a classified claim.
117
+ * A stale claim is only taken over on a clean tree, or when it names the same
118
+ * card (the tree is that card's own unfinished work). No claim + dirty tree is
119
+ * refused: someone who never took a claim is mid-change.
120
+ */
121
+ export function decideAcquire({ classification, claim, card, dirty }) {
122
+ switch (classification.kind) {
123
+ case "own":
124
+ return { write: true, state: "refreshed" };
125
+ case "none":
126
+ return dirty ? { write: false, state: "dirty" } : { write: true, state: "acquired" };
127
+ case "stale":
128
+ return !dirty || claim.card === card
129
+ ? { write: true, state: "took-over" }
130
+ : { write: false, state: "dirty" };
131
+ default:
132
+ // live, corrupt
133
+ return { write: false, state: "held" };
134
+ }
135
+ }
136
+
137
+ // ---------------------------------------------------------------------------
138
+ // git + filesystem
139
+
140
+ function git(cwd, args, env) {
141
+ const result = spawnSync("git", args, { cwd, env, encoding: "utf8" });
142
+ return {
143
+ ok: result.status === 0,
144
+ out: (result.stdout ?? "").trim(),
145
+ err: (result.stderr ?? "").trim(),
146
+ };
147
+ }
148
+
149
+ function readRepo(cwd, env) {
150
+ const gitDir = git(cwd, ["rev-parse", "--absolute-git-dir"], env);
151
+ if (!gitDir.ok) return null;
152
+ const head = git(cwd, ["symbolic-ref", "--short", "-q", "HEAD"], env);
153
+ // --no-optional-locks: never take index.lock, so a status read cannot
154
+ // collide with the holder's own git commands.
155
+ const status = git(cwd, ["--no-optional-locks", "status", "--porcelain"], env);
156
+ return {
157
+ gitDir: gitDir.out,
158
+ head: head.ok && head.out ? head.out : null,
159
+ dirty: status.out.length > 0,
160
+ };
161
+ }
162
+
163
+ function readClaim(path) {
164
+ let raw;
165
+ try {
166
+ raw = readFileSync(path, "utf8");
167
+ } catch (err) {
168
+ if (err?.code === "ENOENT") return null;
169
+ throw err;
170
+ }
171
+ try {
172
+ const parsed = JSON.parse(raw);
173
+ return isClaimRecord(parsed) ? parsed : CORRUPT;
174
+ } catch {
175
+ return CORRUPT;
176
+ }
177
+ }
178
+
179
+ function writeClaim(path, record) {
180
+ const tmp = `${path}.${process.pid}.tmp`;
181
+ writeFileSync(tmp, `${JSON.stringify(record, null, 2)}\n`);
182
+ renameSync(tmp, path);
183
+ }
184
+
185
+ function removeClaim(path) {
186
+ try {
187
+ unlinkSync(path);
188
+ } catch (err) {
189
+ if (err?.code !== "ENOENT") throw err;
190
+ }
191
+ }
192
+
193
+ function sleepSync(ms) {
194
+ Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms);
195
+ }
196
+
197
+ /** Run `fn` inside a mkdir-based critical section; mkdir is atomic everywhere. */
198
+ function withMutex(gitDir, fn) {
199
+ const mutex = join(gitDir, MUTEX_DIR);
200
+ const deadline = Date.now() + MUTEX_RETRY_MS;
201
+ for (;;) {
202
+ try {
203
+ mkdirSync(mutex);
204
+ break;
205
+ } catch (err) {
206
+ if (err?.code !== "EEXIST") throw err;
207
+ try {
208
+ if (Date.now() - statSync(mutex).mtimeMs > MUTEX_STALE_MS) {
209
+ rmSync(mutex, { recursive: true, force: true });
210
+ continue;
211
+ }
212
+ } catch {
213
+ // Vanished between mkdir and stat — retry.
214
+ continue;
215
+ }
216
+ if (Date.now() > deadline) return { busy: true };
217
+ sleepSync(25 + Math.floor(Math.random() * 25));
218
+ }
219
+ }
220
+ try {
221
+ return { value: fn() };
222
+ } finally {
223
+ rmSync(mutex, { recursive: true, force: true });
224
+ }
225
+ }
226
+
227
+ // ---------------------------------------------------------------------------
228
+ // CLI
229
+
230
+ function parseArgs(argv) {
231
+ const opts = { _: [] };
232
+ const valued = new Set(["card", "branch", "session", "ttl", "timeout", "interval"]);
233
+ for (let i = 0; i < argv.length; i += 1) {
234
+ const arg = argv[i];
235
+ if (!arg.startsWith("--")) {
236
+ opts._.push(arg);
237
+ continue;
238
+ }
239
+ const [key, inline] = arg.slice(2).split(/=(.*)/s, 2);
240
+ if (valued.has(key)) {
241
+ const value = inline ?? argv[(i += 1)];
242
+ if (value === undefined || value === "") throw new Error(`--${key} needs a value`);
243
+ opts[key] = value;
244
+ } else if (key === "force" || key === "help") {
245
+ opts[key] = true;
246
+ } else {
247
+ throw new Error(`unknown option --${key}`);
248
+ }
249
+ }
250
+ return opts;
251
+ }
252
+
253
+ function positiveNumber(raw, name) {
254
+ const n = Number(raw);
255
+ if (!Number.isFinite(n) || n <= 0) throw new Error(`--${name} must be a positive number`);
256
+ return n;
257
+ }
258
+
259
+ function describeHolder(claim, classification) {
260
+ if (claim === CORRUPT) return { corrupt: true };
261
+ return {
262
+ ...claim,
263
+ ageSeconds: classification.ageSeconds,
264
+ stale: classification.kind === "stale",
265
+ };
266
+ }
267
+
268
+ function holderHint(holder) {
269
+ if (holder.corrupt)
270
+ return "the claim file does not parse — a human decides: `conveyor-skills checkout release --force`";
271
+ const pid = holder.pid ? ` pid ${holder.pid}` : "";
272
+ return `checkout held by card ${holder.card ?? "?"} on branch ${holder.branch ?? "(detached)"} (${holder.host}${pid}, heartbeat ${holder.ageSeconds}s ago) — do not touch the tree; run \`conveyor-skills checkout wait\``;
273
+ }
274
+
275
+ const HINTS = {
276
+ dirty:
277
+ "working tree is dirty and no live claim covers it — touch nothing; the user commits or stashes",
278
+ "branch-moved": "HEAD is not the claimed branch — do not commit or push; report it",
279
+ lost: "you no longer hold the checkout claim — stop and re-acquire before touching the tree",
280
+ };
281
+
282
+ const COMMANDS = ["acquire", "verify", "release", "status", "wait"];
283
+
284
+ const EXIT_BY_STATE = {
285
+ held: EXIT.held,
286
+ lost: EXIT.held,
287
+ "branch-moved": EXIT.branchMoved,
288
+ dirty: EXIT.dirty,
289
+ };
290
+
291
+ /** Validate parsed options; returns an error message or null. */
292
+ function usageProblem(command, opts) {
293
+ if (!COMMANDS.includes(command)) return `unknown checkout command '${command}'`;
294
+ if (["acquire", "verify"].includes(command) && !opts.card)
295
+ return `${command} needs --card <slug>`;
296
+ if (command === "release" && !opts.card && !opts.force)
297
+ return "release needs --card <slug> (or --force)";
298
+ return null;
299
+ }
300
+
301
+ function acquireOp({ ctx, claim, classification, stamp }) {
302
+ const { opts, repo, identity, hostname, claimPath } = ctx;
303
+ const decision = decideAcquire({ classification, claim, card: opts.card, dirty: repo.dirty });
304
+ if (!decision.write) return { state: decision.state };
305
+ const kept = classification.kind === "own" ? claim : null;
306
+ const record = {
307
+ schema: SCHEMA,
308
+ session: identity.session,
309
+ pid: identity.pid,
310
+ host: hostname,
311
+ card: opts.card,
312
+ branch: opts.branch ?? kept?.branch ?? repo.head,
313
+ acquiredAt: kept?.acquiredAt ?? stamp,
314
+ heartbeatAt: stamp,
315
+ };
316
+ writeClaim(claimPath, record);
317
+ return { state: decision.state, record };
318
+ }
319
+
320
+ function verifyOp({ ctx, claim, classification, stamp }) {
321
+ const { opts, repo, identity, claimPath } = ctx;
322
+ if (classification.kind !== "own") return { state: "lost" };
323
+ const expected = opts.branch ?? claim.branch;
324
+ if (expected && repo.head !== expected) return { state: "branch-moved", expected };
325
+ const record = { ...claim, pid: identity.pid ?? claim.pid, heartbeatAt: stamp };
326
+ writeClaim(claimPath, record);
327
+ return { state: "ok", record };
328
+ }
329
+
330
+ function releaseOp({ ctx, classification }) {
331
+ if (classification.kind === "none") return { state: "none" };
332
+ if (classification.kind !== "own" && !ctx.opts.force) return { state: "held" };
333
+ removeClaim(ctx.claimPath);
334
+ return { state: "released" };
335
+ }
336
+
337
+ const MUTATIONS = { acquire: acquireOp, verify: verifyOp, release: releaseOp };
338
+
339
+ /** Run a mutating command inside the mutex: read, decide, write. */
340
+ function mutate(ctx) {
341
+ // Test-only: widen the read→write window so the parallel-acquire test fails
342
+ // loudly if the mutex ever stops serializing callers.
343
+ const holdMs = Number(ctx.env.CONVEYOR_CHECKOUT_CLAIM_TEST_HOLD_MS) || 0;
344
+ return withMutex(ctx.repo.gitDir, () => {
345
+ const claim = readClaim(ctx.claimPath);
346
+ const classification = ctx.classify(claim);
347
+ if (holdMs > 0) sleepSync(holdMs);
348
+ const stamp = new Date(ctx.now()).toISOString();
349
+ const result = MUTATIONS[ctx.command]({ ctx, claim, classification, stamp });
350
+ return { ...result, claim, classification };
351
+ });
352
+ }
353
+
354
+ /** The JSON line and the stderr hint for a finished mutation. */
355
+ function describeOutcome(base, outcome) {
356
+ const { state, claim, classification, record, expected } = outcome;
357
+ const payload = { ...base, state };
358
+ if (state === "took-over") payload.reason = classification.reason;
359
+ if (record) payload.claim = record;
360
+ if (expected) payload.expectedBranch = expected;
361
+ let hint = HINTS[state] ?? null;
362
+ if (["held", "lost"].includes(state) && claim !== null) {
363
+ payload.holder = describeHolder(claim, classification);
364
+ if (state === "held" || classification.kind === "corrupt") hint = holderHint(payload.holder);
365
+ }
366
+ if (state === "took-over")
367
+ hint = `took over a stale claim (${classification.reason}) from card ${claim.card ?? "?"}`;
368
+ return { payload, hint };
369
+ }
370
+
371
+ /** Parse argv into a command; throws on a usage error. */
372
+ function parseCommand(argv, env) {
373
+ const opts = parseArgs(argv);
374
+ const ttlSeconds = positiveNumber(
375
+ opts.ttl ?? env.CONVEYOR_CHECKOUT_CLAIM_TTL_SECONDS ?? DEFAULT_TTL_SECONDS,
376
+ "ttl",
377
+ );
378
+ return { opts, command: opts._[0], ttlSeconds };
379
+ }
380
+
381
+ /** Everything a command needs to read and judge the claim. */
382
+ function buildContext({ command, opts, ttlSeconds, env, repo, deps }) {
383
+ const now = deps.now ?? (() => Date.now());
384
+ const hostname = deps.hostname ?? osHostname();
385
+ const pidAlive = deps.pidAlive ?? defaultPidAlive;
386
+ const identity = resolveIdentity({ session: opts.session, card: opts.card, env });
387
+ const claimPath = join(repo.gitDir, CLAIM_FILE);
388
+ const classify = (claim) =>
389
+ classifyClaim(claim, { identity, now: now(), hostname, pidAlive, ttlSeconds });
390
+ const base = { command, head: repo.head, dirty: repo.dirty, file: claimPath };
391
+ return { command, opts, env, repo, identity, hostname, claimPath, classify, now, base };
392
+ }
393
+
394
+ async function runStatus({ ctx, emit }) {
395
+ const claim = readClaim(ctx.claimPath);
396
+ const classification = ctx.classify(claim);
397
+ await emit({
398
+ ...ctx.base,
399
+ state: classification.kind,
400
+ ...(classification.reason ? { reason: classification.reason } : {}),
401
+ claim: claim === null ? null : describeHolder(claim, classification),
402
+ });
403
+ return EXIT.ok;
404
+ }
405
+
406
+ async function runMutation({ ctx, emit, hint, fail }) {
407
+ const outcome = mutate(ctx);
408
+ if (outcome.busy) {
409
+ const message = `could not take ${MUTEX_DIR} within ${MUTEX_RETRY_MS}ms — another session is mid-claim; retry`;
410
+ return fail(message, ctx.base);
411
+ }
412
+ const { payload, hint: message } = describeOutcome(ctx.base, outcome.value);
413
+ if (message) hint(message);
414
+ await emit(payload);
415
+ return EXIT_BY_STATE[payload.state] ?? EXIT.ok;
416
+ }
417
+
418
+ /**
419
+ * Run one checkout command. Resolves to the exit code; writes exactly one JSON
420
+ * line to `deps.stdout` and waits for it to flush.
421
+ */
422
+ export async function runCheckoutCli(argv, deps = {}) {
423
+ const env = deps.env ?? process.env;
424
+ const stdout = deps.stdout ?? process.stdout;
425
+ const stderr = deps.stderr ?? process.stderr;
426
+ let command = argv.find((a) => !a.startsWith("-")) ?? null;
427
+ const emit = (payload) =>
428
+ new Promise((resolve) => {
429
+ stdout.write(`${JSON.stringify(payload)}\n`, () => resolve());
430
+ });
431
+ const hint = (msg) => stderr.write(`conveyor-skills checkout: ${msg}\n`);
432
+ const fail = async (message, extra = {}) => {
433
+ hint(message);
434
+ await emit({ command, ...extra, state: "error", error: message });
435
+ return EXIT.error;
436
+ };
437
+
438
+ let parsed;
439
+ try {
440
+ parsed = parseCommand(argv, env);
441
+ } catch (err) {
442
+ return fail(err.message);
443
+ }
444
+ if (parsed.opts.help || !parsed.command) {
445
+ stderr.write(USAGE);
446
+ await emit({ command, state: "usage" });
447
+ return parsed.opts.help ? EXIT.ok : EXIT.error;
448
+ }
449
+ command = parsed.command;
450
+ const problem = usageProblem(command, parsed.opts);
451
+ if (problem) return fail(problem);
452
+
453
+ const repo = readRepo(deps.cwd ?? process.cwd(), env);
454
+ if (!repo) return fail("not inside a git repository");
455
+ const ctx = buildContext({ ...parsed, env, repo, deps });
456
+ const io = { ctx, emit, hint, fail };
457
+ if (command === "status") return runStatus(io);
458
+ if (command === "wait") return runWait(io);
459
+ return runMutation(io);
460
+ }
461
+
462
+ /** Sleep up to `ms`, returning early when `signal.wake()` is called. */
463
+ function interruptibleSleep(ms, signal) {
464
+ return new Promise((resolve) => {
465
+ const timer = setTimeout(resolve, ms);
466
+ signal.wake = () => {
467
+ clearTimeout(timer);
468
+ resolve();
469
+ };
470
+ });
471
+ }
472
+
473
+ const WAIT_VERDICTS = { none: "free", stale: "stale", own: "own" };
474
+
475
+ async function runWait({ ctx, emit, hint, fail }) {
476
+ const { opts, base, claimPath, classify } = ctx;
477
+ let timeoutSeconds;
478
+ let intervalSeconds;
479
+ try {
480
+ timeoutSeconds = positiveNumber(opts.timeout ?? DEFAULT_WAIT_TIMEOUT_SECONDS, "timeout");
481
+ intervalSeconds = positiveNumber(opts.interval ?? DEFAULT_WAIT_INTERVAL_SECONDS, "interval");
482
+ } catch (err) {
483
+ return fail(err.message, base);
484
+ }
485
+
486
+ const signal = { interrupted: false, wake: () => {} };
487
+ const onSignal = () => {
488
+ signal.interrupted = true;
489
+ signal.wake();
490
+ };
491
+ process.once("SIGINT", onSignal);
492
+ process.once("SIGTERM", onSignal);
493
+ const deadline = Date.now() + timeoutSeconds * 1000;
494
+ let announced = false;
495
+
496
+ try {
497
+ for (;;) {
498
+ const claim = readClaim(claimPath);
499
+ const classification = classify(claim);
500
+ const verdict = WAIT_VERDICTS[classification.kind];
501
+ if (verdict) {
502
+ const holder =
503
+ claim && claim !== CORRUPT ? { holder: describeHolder(claim, classification) } : {};
504
+ await emit({ ...base, state: verdict, ...holder });
505
+ return EXIT.ok;
506
+ }
507
+ if (!announced) {
508
+ hint(`waiting — ${holderHint(describeHolder(claim, classification))}`);
509
+ announced = true;
510
+ }
511
+ if (signal.interrupted || Date.now() >= deadline) {
512
+ await emit({ ...base, state: signal.interrupted ? "interrupted" : "timeout" });
513
+ return EXIT.ok;
514
+ }
515
+ await interruptibleSleep(
516
+ Math.min(intervalSeconds * 1000, Math.max(0, deadline - Date.now())),
517
+ signal,
518
+ );
519
+ }
520
+ } finally {
521
+ process.removeListener("SIGINT", onSignal);
522
+ process.removeListener("SIGTERM", onSignal);
523
+ }
524
+ }
package/bin/cli.mjs CHANGED
@@ -7,6 +7,9 @@
7
7
  // Usage:
8
8
  // conveyor-skills link (default) create/refresh links, prune stale ones
9
9
  // conveyor-skills link --check verify links; exit 1 if missing/stale
10
+ // conveyor-skills checkout <acquire|verify|release|status|wait> ...
11
+ // take turns on a shared local checkout
12
+ // (see bin/checkout-claim.mjs)
10
13
  //
11
14
  // Only links owned by this package (target path contains conveyor-skills/skills/)
12
15
  // are ever replaced or pruned. Real directories and foreign symlinks are left
@@ -68,10 +71,17 @@ function ownedBy(linkPath) {
68
71
  }
69
72
 
70
73
  const args = process.argv.slice(2);
74
+
75
+ if (args[0] === "checkout") {
76
+ const { runCheckoutCli } = await import("./checkout-claim.mjs");
77
+ // runCheckoutCli resolves only after its JSON line has flushed.
78
+ process.exit(await runCheckoutCli(args.slice(1)));
79
+ }
80
+
71
81
  const checkOnly = args.includes("--check");
72
82
  const command = args.find((a) => !a.startsWith("-")) ?? "link";
73
83
  if (command !== "link")
74
- fail(`unknown command '${command}' — only 'link' (with optional --check) exists`);
84
+ fail(`unknown command '${command}' — expected 'link' (with optional --check) or 'checkout'`);
75
85
 
76
86
  // This package's own skills/ dir, resolved through any install symlink (bun
77
87
  // may install the package as a symlink into its store).
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@rallycry/conveyor-skills",
3
- "version": "1.0.12",
3
+ "version": "1.0.13",
4
4
  "description": "Shared Claude Code skills for Conveyor consumer repos, linked into .claude/skills via the conveyor-skills CLI",
5
5
  "keywords": [
6
6
  "claude",
@@ -20,5 +20,9 @@
20
20
  "type": "module",
21
21
  "publishConfig": {
22
22
  "access": "public"
23
+ },
24
+ "scripts": {
25
+ "test": "node --test __tests__/checkout-claim.test.mjs",
26
+ "test:unit": "node --test __tests__/checkout-claim.test.mjs"
23
27
  }
24
28
  }
@@ -102,6 +102,12 @@ task binding and no user account.
102
102
  stack) is the entire value of local execution, and worktrees have none of it.
103
103
  A dirty `git status` before any branch switch is a hard stop: report and let
104
104
  the user commit or stash. Do not solve it with a second checkout.
105
+ - **Take turns on the checkout.** Another local session can share this
106
+ checkout, and a clean tree does not prove it is idle. Hold the checkout
107
+ claim from before the card claim until the finish line, and never switch
108
+ branches while another session holds it — the protocol, commands, and what
109
+ each result means are in
110
+ [references/checkout-claim.md](references/checkout-claim.md). A pod skips it.
105
111
  - The dev database and dev-server ports are shared with the user's own
106
112
  sessions. No destructive experiments, never reset the dev DB, and reuse a
107
113
  running dev stack rather than fighting over ports.
@@ -149,10 +155,13 @@ say why.
149
155
 
150
156
  ## Claim
151
157
 
152
- Re-confirm the card is still claimable, then `mcp__conveyor__update_task` →
158
+ Re-confirm the card is still claimable. **Locally**, take the checkout claim
159
+ first (`checkout acquire --card <slug>`,
160
+ [references/checkout-claim.md](references/checkout-claim.md)); `held` means
161
+ flag and wait, not claim. Then `mcp__conveyor__update_task` →
153
162
  `status: "InProgress"`, then `mcp__conveyor__post_to_chat` with a claim marker
154
163
  naming where you are running. Status changed under you → someone else took it;
155
- stop.
164
+ release the checkout claim and stop.
156
165
 
157
166
  **A pack parent is claimed differently** — see the pack path. The parent stays
158
167
  parked and the chat marker IS the claim.
@@ -285,7 +294,10 @@ Refresh against the card's base first, then gate, then open — in that order, s
285
294
  nothing invalidates the verification you just did:
286
295
 
287
296
  1. `git fetch origin <base> && git merge origin/<base> --no-edit`
288
- 2. One verification pass, scoped to the diff.
297
+ 2. One verification pass, scoped to the diff. **Locally**, run
298
+ `checkout verify --card <slug>` before the pass, each commit and push, and
299
+ the PR call; anything but `ok` means stop (see
300
+ [references/checkout-claim.md](references/checkout-claim.md)).
289
301
  3. `mcp__conveyor__create_pull_request`, naming the base branch **explicitly**.
290
302
  **The two surfaces spell these differently and unknown keys are silently
291
303
  dropped, not rejected** — locally it is `head:` / `base:`; in a pod it is
@@ -298,6 +310,8 @@ nothing invalidates the verification you just did:
298
310
  should look at, and anything you did NOT do.
299
311
 
300
312
  Then confirm CI actually started (read-only `gh pr checks`). Do not wait on it.
313
+ **Locally**, that is the finish line for the checkout too:
314
+ `checkout release --card <slug>`.
301
315
 
302
316
  **Do not re-merge the base and do not re-run a gate that already passed.** If
303
317
  the base moved while the gates ran, open the PR anyway — CI validates against
@@ -332,8 +346,9 @@ After two genuinely different failed approaches, or on a decision only the user
332
346
  can make: post the reason AND the specific question to chat, set the card back
333
347
  to `"Open"`, restore the tree (locally, `git checkout dev` — leaving a shared
334
348
  checkout parked on an abandoned feature branch is how the next session starts
335
- from the wrong base), and stop. A vague "this is hard" is not a handoff; the
336
- question is what makes it one.
349
+ from the wrong base), release the checkout claim locally
350
+ (`checkout release --card <slug>`), and stop. A vague "this is hard" is not a
351
+ handoff; the question is what makes it one.
337
352
 
338
353
  **A thin plan is a different case, and it does not stop a pack.** A card whose
339
354
  plan fails the context-free-reader bar: post what is missing to its chat and
@@ -0,0 +1,98 @@
1
+ # Checkout claim — local sessions take turns on one checkout
2
+
3
+ The protocol behind every "hold the checkout claim" line in
4
+ [conveyor-build](../SKILL.md), its [task path](task-path.md) and
5
+ [pack path](pack-path.md), [conveyor-local-loop](../../conveyor-local-loop/SKILL.md),
6
+ and local fixes in [conveyor-review](../../conveyor-review/SKILL.md).
7
+
8
+ **Why.** The card claim (`InProgress` plus a chat marker) stops two sessions
9
+ working ONE card. It does nothing about two sessions working two cards in ONE
10
+ checkout. A clean tree passes the "dirty `git status` → stop" rule even while
11
+ another session is mid-task, so `git checkout -B <branch>` succeeds and switches
12
+ the branch underneath it. The checkout claim closes that gap: one session holds
13
+ the working tree at a time, and every other session waits.
14
+
15
+ **Scope: local only.** A pod has its own checkout and never runs this. Skip
16
+ every checkout-claim step when you are in a pod.
17
+
18
+ ## How to call it
19
+
20
+ The helper is the `checkout` subcommand of the `conveyor-skills` bin. Pick the
21
+ form once per session:
22
+
23
+ ```bash
24
+ [ -x node_modules/.bin/conveyor-skills ] && echo local || echo npx
25
+ ```
26
+
27
+ - `local` → `node_modules/.bin/conveyor-skills checkout <command> …`
28
+ - `npx` → `npx -y -p @rallycry/conveyor-skills@latest conveyor-skills checkout <command> …`
29
+ - The local bin answers `unknown command 'checkout'` (an older branch or an old
30
+ install) → use the `npx` form.
31
+
32
+ Write the chosen form out literally in every call. Never store it in a shell
33
+ variable and run `$VAR checkout …`: zsh does not word-split it.
34
+
35
+ `--card` is the card you are working. Inside a pack it is ALWAYS the pack
36
+ parent's slug, for the whole run. Identity comes from `CONVEYOR_SESSION_ID`, then
37
+ `CLAUDE_CODE_SESSION_ID`. A runtime that sets neither passes a stable
38
+ `--session <id>` on every call; without it the identity falls back to
39
+ `card:<slug>`, which still works but frees a dead holder only after the TTL.
40
+
41
+ The claim is one file, `<git dir>/conveyor-checkout-claim.json`. It is never
42
+ committed and never shows in `git status`. Each command prints ONE line of
43
+ JSON on stdout (`{"command","state","head","dirty",…}`) and a hint on stderr.
44
+
45
+ ## When
46
+
47
+ | Moment | Command |
48
+ | --- | --- |
49
+ | Before the first git write or card status write (before the claiming `update_task`) | `acquire --card <slug>` |
50
+ | After every branch create or switch | `acquire --card <slug> --branch <branch>` |
51
+ | Before each gate, commit, push, and `create_pull_request`; at the start of every wake while you hold the claim | `verify --card <slug>` (also refreshes the heartbeat) |
52
+ | Finish line reached (PR open, CI started), parked, blocked, or the card claim lost — after restoring the tree | `release --card <slug>` |
53
+
54
+ ## What each result means
55
+
56
+ | Command | State (exit) | Do this |
57
+ | --- | --- | --- |
58
+ | `acquire` | `acquired`, `refreshed` (0) | Carry on. |
59
+ | `acquire` | `took-over` (0) | The previous holder was dead or stale. If `dirty` is true, the tree holds that claim's unfinished work for THIS card: audit it before you continue. |
60
+ | `acquire` | `held` (3) | Another session holds the checkout. Touch nothing — no checkout, no status write, no commit. Flag it (below), then wait. |
61
+ | `acquire` | `dirty` (5) | The tree is dirty and no live claim covers it. Hard stop: report it. The user commits or stashes, never you. |
62
+ | `verify` | `ok` (0) | Carry on. |
63
+ | `verify` | `branch-moved` (4) | HEAD is not the claimed branch — someone switched it, or your own checkout failed and left you on the previous branch. Do not commit or push. Report it and stop. |
64
+ | `verify` | `lost` (3) | You no longer hold the claim. Stop touching the tree and `acquire` again, then follow that result. |
65
+ | `release` | `released`, `none` (0) | Done. |
66
+ | `release` | `held` (3) | The claim is not yours. Leave it. |
67
+ | any | `error` (1) | Usage error or not a git repo. Read the stderr line. |
68
+
69
+ **Flag, then wait.** On `held`, tell the user which card, branch and host hold
70
+ the checkout (the `holder` field has all three). A standalone build also posts
71
+ one line to its card's chat, for example
72
+ `[build] waiting — checkout held by <card> on <branch>`. Then wait:
73
+
74
+ ```bash
75
+ node_modules/.bin/conveyor-skills checkout wait --timeout 1740 # or the npx form
76
+ ```
77
+
78
+ In Claude Code, launch it with Bash `run_in_background: true` and end the
79
+ turn. Its completion notification is the wake. In Codex, hold the command
80
+ session open until it returns. The result is advisory, like `conveyor-wait`:
81
+ `free`, `stale`, or `own` → run `acquire` again; `timeout` → arm ONE new wait.
82
+ Never arm a second wait while one is running.
83
+
84
+ ## Rules that go with the claim
85
+
86
+ - **Stage by path, never `git add -A`.** In a shared checkout an untracked file
87
+ can belong to another session.
88
+ - **No claim, dirty tree → refuse.** `acquire` enforces the old "dirty tree is
89
+ a hard stop" rule deterministically. It also covers ad-hoc sessions that
90
+ never take a claim.
91
+ - **A stale claim is taken over only on a clean tree, or when it names the
92
+ same card.** A holder is stale when its pid is dead on this host, when the
93
+ same pid now runs a different session (a `/clear`), or when its heartbeat is
94
+ older than the TTL (7200s; `CONVEYOR_CHECKOUT_CLAIM_TTL_SECONDS` or `--ttl`
95
+ overrides it).
96
+ - **Human override.** `checkout status` shows the holder and never changes
97
+ anything. `checkout release --force` drops someone else's claim, or a claim
98
+ file that does not parse. Run it only when the user tells you to.
@@ -24,6 +24,13 @@ alternative is where pack incidents come from.
24
24
  IMMEDIATELY record it: `mcp__conveyor__update_task` with
25
25
  `githubBranch: <branch>`.
26
26
 
27
+ **Locally, take the checkout claim before any of this** —
28
+ `checkout acquire --card <parent-slug>` before you cut or check out the
29
+ pack branch, then `acquire --card <parent-slug> --branch <pack>` once you
30
+ are on it ([checkout-claim.md](checkout-claim.md)). The pack holds the
31
+ claim for the WHOLE run: `--card` stays the parent's slug through every
32
+ child, and `--branch` follows each child-branch and pack-branch switch.
33
+
27
34
  That write is load-bearing, not bookkeeping. Identification mints a
28
35
  competing `conveyor/*` branch name onto any branchless card it processes,
29
36
  and every pack-child merge handler keys on the card's recorded branch
@@ -121,6 +128,10 @@ state and is a no-op. This is NOT the "insurance wakeup" the pod prompt
121
128
  forbids: that rule is about a background job's own completion notification,
122
129
  which does fire. Nothing at all notifies a pack between children, so the
123
130
  heartbeat is the only guaranteed wake a Claude pack has. Do not delete it.
131
+ Locally, every wake — heartbeat or not — starts with
132
+ `checkout verify --card <parent-slug>` before it touches the tree, and a
133
+ result other than `ok` stops the pack (see
134
+ [checkout-claim.md](checkout-claim.md)).
124
135
 
125
136
  **Codex CLI.** There is no `ScheduleWakeup` and no `/loop`. The thread goal
126
137
  created per the SKILL.md *Goal and finish line* section is the continuation:
@@ -153,7 +164,10 @@ parent chat, skip that child, and take the next ready one.
153
164
  Then follow [task-path.md](task-path.md), with one substitution: the child
154
165
  branches from the **pack branch**, not `dev`, and its PR's base is the pack
155
166
  branch. Name that base explicitly — the default is `dev`, and the two surfaces
156
- spell the argument differently (`base:` locally, `baseBranch:` in a pod).
167
+ spell the argument differently (`base:` locally, `baseBranch:` in a pod). The
168
+ pack already holds the checkout claim, so the task path's claim steps become
169
+ `acquire --card <parent-slug> --branch <child-branch>` after you cut the child
170
+ branch, and there is no release at the child's finish line.
157
171
 
158
172
  > **Environment — a pod has no child branches and no child PRs.** Everything
159
173
  > above describes the LOCAL model, where each child gets its own branch and its
@@ -261,7 +275,8 @@ leave the branch pushed, and stop.
261
275
  `base:` `dev`. The parent moves to ReviewPR. Post the pack summary: what
262
276
  shipped per child, how it was verified, what reviewers should look at.
263
277
  4. Confirm CI started. **Never approve or merge this PR** — it is the one that
264
- gets independent review.
278
+ gets independent review. Locally, release the checkout:
279
+ `checkout release --card <parent-slug>`.
265
280
 
266
281
  ## Parked protocol
267
282
 
@@ -269,4 +284,6 @@ After two genuinely different failed approaches on a child, or a decision only
269
284
  the user can make: post the reason and the specific question to the child AND
270
285
  parent chats, set the child back to `Open`, restore the tree, and take the next
271
286
  child whose dependency chain does not run through the parked one. Everything
272
- remaining blocked → report and stop; the user's reply is the un-park signal.
287
+ remaining blocked → restore the tree, release the checkout claim locally
288
+ (`checkout release --card <parent-slug>`), report, and stop; the user's reply
289
+ is the un-park signal, and the resumed run acquires again.
@@ -13,11 +13,14 @@ Re-confirm via `mcp__conveyor__get_task` that the card is still claimable —
13
13
  `Open` (or whatever status the user explicitly overrode), no assignee you do
14
14
  not expect, no active session. Then:
15
15
 
16
- 1. `mcp__conveyor__update_task` → `status: "InProgress"`
17
- 2. `mcp__conveyor__post_to_chat` → `[build] claimed — <where you are running>`
16
+ 1. **Locally only:** `checkout acquire --card <slug>` — the checkout claim from
17
+ [checkout-claim.md](checkout-claim.md). `held` → touch nothing, flag it, and
18
+ wait; `dirty` → hard stop. A pod skips this step.
19
+ 2. `mcp__conveyor__update_task` → `status: "InProgress"`
20
+ 3. `mcp__conveyor__post_to_chat` → `[build] claimed — <where you are running>`
18
21
 
19
22
  If the status changed under you between the read and the write, someone else
20
- took it. Stop rather than compete.
23
+ took it. Release the checkout claim and stop rather than compete.
21
24
 
22
25
  ## 2. Branch from the card's base — never blindly `dev`
23
26
 
@@ -29,6 +32,10 @@ git fetch origin <base>
29
32
  git checkout -B <feat|fix|chore>/<slug> origin/<base>
30
33
  ```
31
34
 
35
+ Locally, record the branch on the checkout claim straight after the checkout:
36
+ `checkout acquire --card <slug> --branch <feat|fix|chore>/<slug>`. Do the
37
+ same after any later branch switch.
38
+
32
39
  Two cases that are not a fresh branch:
33
40
 
34
41
  - **The card already has a `githubBranch` with commits on origin.** Resume THAT
@@ -39,7 +46,8 @@ Two cases that are not a fresh branch:
39
46
  absent even though the branch exists on the remote. Fetch it explicitly
40
47
  (`git fetch origin <branch>`) rather than concluding the branch is missing —
41
48
  and note that a failed `checkout` leaves you on your PREVIOUS branch, where a
42
- follow-up `git push` will push the wrong thing.
49
+ follow-up `git push` will push the wrong thing. Locally, `checkout verify`
50
+ reports exactly that as `branch-moved`.
43
51
 
44
52
  Reinstall dependencies if the lockfile changed.
45
53
 
@@ -62,6 +70,8 @@ Chat at milestones, not per step.
62
70
  These are identical to the SKILL.md sections of the same names. In particular:
63
71
  sync the base BEFORE the verification pass, pass `base:` explicitly to
64
72
  `mcp__conveyor__create_pull_request`, and confirm CI started without waiting on
65
- it.
73
+ it. Locally, `checkout verify --card <slug>` before the gate, each commit and
74
+ push, and the PR call.
66
75
 
67
76
  Finish line: the card in ReviewPR with CI started. You do not merge it.
77
+ Locally, `checkout release --card <slug>` once CI has started.
@@ -47,12 +47,27 @@ adds only selection, claiming, cadence, and local-machine hygiene.
47
47
  Clean tree before any branch switch is still a hard rule: dirty
48
48
  `git status` at iteration start → touch nothing, report, and idle — the
49
49
  resolution is the user committing or stashing, not a second checkout.
50
+ - **Hold the checkout claim while you use the tree.** WIP = 1 is per loop,
51
+ not per checkout: two loops, or a loop plus an interactive session, can
52
+ share this checkout, and a clean tree does not prove it is idle. Every tier
53
+ that touches git takes the claim first and gives it back at the finish line —
54
+ protocol in [conveyor-build's checkout-claim reference](../conveyor-build/references/checkout-claim.md).
50
55
  - **Push early.** There is no pod WIP-autosync locally; committed-and-pushed is
51
56
  the only durable state. Push the branch (`-u origin`) as soon as it exists.
52
57
 
53
58
  ## Iteration order
54
59
 
55
- Each invocation does the FIRST of these that produces work, then paces:
60
+ Each invocation does the FIRST of these that produces work, then paces.
61
+
62
+ **Checkout first.** Start every iteration with `checkout status`. `live` or
63
+ `corrupt` means another session holds this checkout: skip Recover, Babysit
64
+ fixes, and Claim (all three touch the tree), arm ONE background
65
+ `checkout wait` (see *Waking on board events*), and pace. Otherwise, each tier
66
+ below runs `checkout acquire --card <slug>` for its card before its first git
67
+ command or status write; `held` → stop that tier and arm the wait. The chat
68
+ marker names only the host, so a second loop on this host also matches
69
+ "my" marker in Recover — the checkout claim is what keeps it off a card another
70
+ live loop is still working.
56
71
 
57
72
  1. **Recover** — a card with my `[local-loop] claimed` chat marker still
58
73
  InProgress without a PR? Resume it. The branch may already exist locally or
@@ -89,13 +104,14 @@ Each invocation does the FIRST of these that produces work, then paces:
89
104
  Skip `followParentStatus` mirror children. A blocker counts as met only
90
105
  when merged-or-beyond (ReviewDev/ReviewLive/Complete) or Cancelled — a
91
106
  blocker sitting in ReviewPR is NOT met until its PR merges.
92
- 2. Claim: re-confirm via `get_task` it is still Open, then
93
- `mcp__conveyor__update_task` → `status: "InProgress"`, then
107
+ 2. Claim: re-confirm via `get_task` it is still Open, take the checkout claim
108
+ (`checkout acquire --card <slug>`; `held` → do not claim, arm the wait),
109
+ then `mcp__conveyor__update_task` → `status: "InProgress"`, then
94
110
  `mcp__conveyor__post_to_chat`: `[local-loop] claimed — working locally on
95
111
  <hostname>`. Claiming a PACK is different — never set the parent
96
112
  InProgress: leave it Open and post the pack claim marker instead (see Pack
97
- mode). Status changed under you → someone else took it; next
98
- candidate.
113
+ mode). Status changed under you → someone else took it; release the
114
+ checkout claim and take the next candidate.
99
115
  3. Plan missing or failing the context-free-reader bar → don't wing it: post
100
116
  what's missing to chat, leave the card Open, skip it.
101
117
 
@@ -115,15 +131,18 @@ This section adds only what the LOOP changes:
115
131
  re-check the parent's status; if it went InProgress or ReviewPR, post
116
132
  `[local-loop] parked: pack coordinator active — yielding`, leave the branch
117
133
  pushed, and let the coordinator take over.
118
- - **Do not wait on CI.** Confirm it started (read-only `gh pr checks`) and end
119
- the iteration — the Babysit tier owns it from there. This is the loop's one
120
- real departure from a standalone build, which stays with its PR.
134
+ - **Do not wait on CI.** Confirm it started (read-only `gh pr checks`),
135
+ `checkout release --card <slug>`, and end the iteration — the Babysit tier
136
+ owns it from there. This is the loop's one real departure from a standalone
137
+ build, which stays with its PR. A Babysit fix re-acquires with that card and
138
+ releases again once the fix is pushed. A pack child's PR is the exception:
139
+ the pack keeps the claim until its finale PR opens.
121
140
 
122
141
  **Parked protocol** — after 2 genuinely different failed approaches, or on a
123
142
  decision only the user can make: post `[local-loop] parked: <reason + the
124
143
  specific question>`, set status back to `"Open"`, restore the tree
125
- (`git checkout dev`), move on. The user's next chat reply is the un-park
126
- signal.
144
+ (`git checkout dev`), `checkout release --card <slug>`, move on. The user's
145
+ next chat reply is the un-park signal.
127
146
 
128
147
  ## Pack mode
129
148
 
@@ -136,9 +155,9 @@ dev→pack sync after every merge, cross-reference, then the finale parent PR
136
155
  into dev. That reference is the source of truth for the procedure; this section
137
156
  only defines how it embeds in the loop:
138
157
 
139
- - The pack occupies the loop's single WIP slot from claim until the finale PR
140
- opens. Do not interleave unrelated cards mid-pack — that thrashes branch
141
- state.
158
+ - The pack occupies the loop's single WIP slot — and holds the checkout
159
+ claim under the parent's slug — from claim until the finale PR opens. Do not
160
+ interleave unrelated cards mid-pack — that thrashes branch state.
142
161
  - Claim marker goes to the PARENT chat: `[local-loop] claimed — driving this
143
162
  pack locally on <hostname>, pack branch <branch>`. The parent's status
144
163
  stays Open (parked); the marker is the claim.
@@ -208,9 +227,12 @@ On wake, run a normal iteration and re-enumerate with
208
227
  `mcp__conveyor__list_tasks`. By then the card may be claimed, cancelled, or
209
228
  blocked by a dependency — never claim straight from the wait payload.
210
229
 
211
- **Never arm a second wait.** On a wake where a wait process is still in flight
212
- and the queue is still empty, re-arm the fallback `ScheduleWakeup` and end the
213
- turn.
230
+ **Never arm a second wait.** At most ONE background wait of either kind —
231
+ `conveyor-wait` or `checkout wait` — runs at a time. On a wake where a wait
232
+ process is still in flight and nothing else changed, re-arm the fallback
233
+ `ScheduleWakeup` and end the turn. When the checkout is held, the
234
+ `checkout wait` is the one to arm: the board wait cannot tell you when the
235
+ checkout frees up.
214
236
 
215
237
  ## Pacing (dynamic /loop only)
216
238
 
@@ -219,7 +241,7 @@ Under `/loop` with no interval, end EVERY iteration with exactly one
219
241
 
220
242
  | State | Delay | Reason should say |
221
243
  |-------|-------|-------------------|
222
- | A background gate/agent/conveyor-wait is in flight — its completion notification is the real wake | 1200–1800s fallback | "fallback while <gate> runs — its notification wakes me sooner" |
244
+ | A background gate/agent/conveyor-wait/checkout wait is in flight — its completion notification is the real wake | 1200–1800s fallback | "fallback while <gate> runs — its notification wakes me sooner" |
223
245
  | ANY actionable work exists: claimable cards or packs, a pack child to implement/merge, a PR still to open, red/pending CI, review comments | 60–90s | queue depth / which item is next |
224
246
  | Queue enumerated as empty THIS iteration, all loop PRs green and quiet | 1200–1800s | queue empty; conveyor-wait armed, so this is only the fallback |
225
247
  | Loop-fatal: MCP dead after 2 tries, dirty tree, broken repo | notify the user (PushNotification if available), then 1800s — or `stop: true` if continuing is unsafe | what is wrong |
@@ -249,7 +271,8 @@ depth each iteration so the user can offload manually.
249
271
  - Not a pod: no sandbox, no WIP snapshots, and the dev DB + dev-server ports
250
272
  are shared with the user's interactive sessions — no destructive
251
273
  experiments, never reset the dev DB, reuse a running dev stack rather than
252
- fighting over ports.
274
+ fighting over ports. The git checkout is shared the same way, which is why
275
+ the loop takes turns on it through the checkout claim.
253
276
  - Not a parallel executor: one card (or one pack) at a time is the point (the
254
277
  full machine per gate). Backlogged? That is what claudespaces — or the
255
278
  offload valve — are for.
@@ -25,10 +25,11 @@ most two packs.
25
25
  **Finish line:** the report — verdict on its first line — posted to the
26
26
  conversation, and to the release card's chat when a release card exists.
27
27
 
28
- **A turn may end only when:** (a) the finish line is true, or (b) an
29
- `AskUserQuestion` is pending because the release scope is genuinely ambiguous.
30
- Anything else — a ledger with blank rows, findings listed but not filed, packs
31
- created with no report — is a stalled run, not a paused one.
28
+ **A turn may end only when:** (a) the finish line is true, (b) an
29
+ `AskUserQuestion` is pending because the release scope is genuinely ambiguous,
30
+ or (c) agents you launched are still running; their completion notices resume
31
+ the run. Anything else — a ledger with blank rows, findings listed but not
32
+ filed, packs created with no report — is a stalled run, not a paused one.
32
33
 
33
34
  **Runtime:** *Codex CLI* — before step 0, create a thread goal with your goal
34
35
  tool: "Post a release verdict for <release> with every change evaluated and
@@ -63,6 +64,9 @@ the finish line above is the loop.
63
64
  access and any release checklist live in the host repo's CLAUDE.md. A wrapper
64
65
  skill may add dimensions (see the end of this file); it cannot lower the
65
66
  evidence bar.
67
+ - **The fan-out rules are not project policy.** They follow from how the
68
+ harness behaves and from account limits, so no project can relax them. A
69
+ wrapper may LOWER the concurrency cap; it may never raise it or lift a rule.
66
70
 
67
71
  ## 0 — Discover the project
68
72
 
@@ -131,9 +135,21 @@ Anchor to a concrete diff before reading any code.
131
135
  5. **Build the coverage ledger** — one row per PR or card, one row per
132
136
  dimension in step 2. Every row ends `clean`, `finding`, or
133
137
  `not reviewable (why)`. "Evaluate everything" means the ledger has no blank
134
- rows when you finish. On a large release, fan subagents out per area and
135
- give each the ground rules and the evidence bar verbatim; the ledger is
136
- still the completeness check, and it stays with you.
138
+ rows when you finish.
139
+
140
+ **Fanning out.** On a large release, split the ledger across subagents per
141
+ area. Give each the ground rules and the evidence bar verbatim. Three rules
142
+ hold, and the mechanics are in [references/fan-out.md](references/fan-out.md):
143
+
144
+ 1. **The hand-back is the only report channel.** The harness refuses report
145
+ files written by subagents. Each agent's final message is its report, and
146
+ you save each one to the run directory the moment it arrives.
147
+ 2. **Only you spawn agents.** Every agent prompt forbids the Agent, Task and
148
+ Workflow tools. Nested agents share your session limit.
149
+ 3. **At most four agents in flight**, the adversarial agent included. Launch
150
+ the next slice when one hands back, never in fixed waves.
151
+
152
+ The ledger is still the completeness check, and it stays with you.
137
153
 
138
154
  ## 2 — Review every dimension
139
155
 
@@ -206,9 +222,12 @@ in it that mattered.
206
222
  ## 5 — The adversarial pass
207
223
 
208
224
  Every candidate that reached step 4, **together with the fix you would
209
- propose**, now faces a reviewer whose only job is to strike it. Use a fresh
210
- subagent given the finding, its evidence and the tests below, told to argue
211
- for striking. Where subagents are unavailable, switch stance explicitly and
225
+ propose**, now faces a reviewer whose only job is to strike it. Use one fresh
226
+ subagent for all candidates, under the fan-out rules. Write the candidates,
227
+ with their evidence, priority and proposed fix, to `candidates.md` in the run
228
+ directory; with no fan-out, put them in the prompt instead. The agent argues
229
+ for striking each against the tests below and hands back one verdict per
230
+ candidate. Where subagents are unavailable, switch stance explicitly and
212
231
  write the case against each finding before deciding.
213
232
 
214
233
  Strike when:
@@ -344,6 +363,10 @@ moves.
344
363
  already happened.
345
364
  - Diagnosing an incident here. An incident the release implicates is
346
365
  evidence; its root cause belongs to `/conveyor-triage`.
366
+ - Telling subagents to write findings files. The harness refuses them. The
367
+ hand-back is the report, and saving it is your job.
368
+ - Letting a reviewer fan out. Its agents share your session limit, and a limit
369
+ hit kills every agent in flight with its unsaved work.
347
370
 
348
371
  ## Wrapping this skill for one project
349
372
 
@@ -364,6 +387,10 @@ evidence bar and adversarial pass.
364
387
  re-check before concluding.
365
388
  - The log tools appear only for a project with the matching integration. If
366
389
  they are absent, the project has none; say so instead of hunting.
390
+ - `You've hit your session limit · resets <time>` (HTTP 429) is the account's
391
+ five-hour usage window, shared by every agent and by you. Nothing runs until
392
+ the reset, so do not retry. Resume per
393
+ [references/fan-out.md](references/fan-out.md).
367
394
 
368
395
  ## Improve This Skill
369
396
 
@@ -287,3 +287,9 @@ Evidence tools:
287
287
 
288
288
  A wrapper may add dimensions and name tools. It may not lower the evidence
289
289
  bar, skip the adversarial pass, or change what the priority levels mean.
290
+
291
+ A wrapper also may not relax the fan-out rules in
292
+ [fan-out.md](fan-out.md): subagents hand back text and write no report files,
293
+ only the orchestrator spawns agents, and at most four run at once. These
294
+ follow from how the harness behaves and from account limits, not from project
295
+ policy. A wrapper may LOWER the cap; it may never raise it or lift a rule.
@@ -0,0 +1,130 @@
1
+ # Fanning out — the mechanics
2
+
3
+ How the orchestrator of a [release review](../SKILL.md) splits the ledger
4
+ across subagents without losing their work. SKILL.md states the three rules;
5
+ this file is how to follow them.
6
+
7
+ ## Why
8
+
9
+ Two limits broke earlier runs. Neither is project policy. Both come from how
10
+ the harness and the account behave, so a wrapper skill may lower the cap
11
+ below but never raise it or lift a rule.
12
+
13
+ 1. **The harness refuses report files from subagents.** Claude Code's Write
14
+ tool refuses a subagent `.md` whose name starts with REPORT, SUMMARY,
15
+ FINDINGS or ANALYSIS:
16
+
17
+ ```
18
+ Subagents should return findings as text, not write report files.
19
+ ```
20
+
21
+ This guard is deliberate. Do not route around it with another file name or
22
+ a Bash heredoc. The subagent's final message is the report.
23
+
24
+ 2. **Every agent draws on one account session limit.** The limit is a single
25
+ five-hour usage window. Every subagent and the orchestrator share it. When
26
+ it runs out, every call fails at once:
27
+
28
+ ```
29
+ You've hit your session limit · resets 9:20pm
30
+ ```
31
+
32
+ (HTTP 429, `rateLimitType: "five_hour"`). Every agent in flight dies, and
33
+ its unsaved work dies with it. The orchestrator is down too, until the
34
+ reset. A run with 13 agents in flight lost 10; a run with about 10 in flight
35
+ (one reviewer spawned 3 of its own) lost 5. Runs at 7 and at 4 lost none.
36
+
37
+ No cap makes a run safe, because the limit is account-wide. The cap bounds
38
+ what one hit destroys. The no-nesting rule and the tool-call budget cut what a
39
+ run spends.
40
+
41
+ ## The run directory
42
+
43
+ `/tmp/conveyor-release-review/<label-slug>/`, where `<label-slug>` is the
44
+ release label in kebab case (`release-2026-09-30`, `pre-release-2026-09-30`).
45
+ Create it before the first launch. It holds:
46
+
47
+ - `brief.md` — the shared part of every agent prompt: the brief block below,
48
+ the ground rules and the evidence bar, verbatim.
49
+ - `ledger.md` — the coverage ledger. Its first line names the run directory,
50
+ so a resumed or compacted session finds it again.
51
+ - `<slice>.md` — one file per agent, holding that agent's hand-back.
52
+
53
+ **On each completion notice, in this order:**
54
+
55
+ 1. Write the hand-back verbatim to `<slice>.md`. Do this before you read it
56
+ closely, and before anything else can fail.
57
+ 2. Mark the ledger rows the hand-back covers.
58
+ 3. Launch the next slice, if one is waiting.
59
+
60
+ The saved files are the durable record. They survive compaction and a resume,
61
+ and the adversarial agent reads findings from them by path.
62
+
63
+ ## The brief block
64
+
65
+ Paste this verbatim into every agent prompt, area and adversarial alike:
66
+
67
+ ```markdown
68
+ 1. Your final message is your report. Do not write report, summary or
69
+ findings files, and do not route a refused write through Bash or another
70
+ file name. The orchestrator saves your final message.
71
+ 2. Do not spawn agents or run workflows (no Agent, Task or Workflow tool).
72
+ The orchestrator owns all fan-out.
73
+ 3. Budget: about 150 tool calls. At the budget, stop and return what you
74
+ have, listing what you did not reach as "not reviewed".
75
+ 4. Already settled, do not re-derive: <list, or "nothing yet">.
76
+ ```
77
+
78
+ Then give the hand-back format:
79
+
80
+ - **Coverage** — one row per PR or card: `clean`, `finding`, or
81
+ `not reviewable (why)`.
82
+ - **Findings** — each with its priority, the PR that introduced it, the
83
+ evidence grade, the exact command and its result, the smallest fix, and the
84
+ strongest case for striking it.
85
+ - **Unverified** — each with the query that would settle it.
86
+ - **Struck** — each with its reason.
87
+ - **Deploy-checklist items.**
88
+
89
+ ## Scheduling
90
+
91
+ - **At most 4 agents in flight**, the adversarial agent included.
92
+ - **Launch as slots free up**, not in fixed waves. When one agent hands back
93
+ and its file is saved, launch the next slice.
94
+ - **One adversarial agent.** Launch it only after every area file exists and
95
+ you have finished steps 3 and 4. First write `candidates.md`: every
96
+ candidate that reached step 4, with its evidence, its priority and the fix
97
+ you would propose. The area files hold raw hand-backs, so they miss your own
98
+ candidates, your grading and your fixes. The agent reads `candidates.md`,
99
+ opens the area files only for detail, argues for striking each candidate,
100
+ and hands back one verdict per candidate. Save its hand-back to
101
+ `adversarial.md`.
102
+ - While agents run, end the turn. Their completion notices resume the run.
103
+
104
+ ## When an agent dies
105
+
106
+ **A session limit** (`session limit · resets <time>`):
107
+
108
+ 1. You are down too. Nothing runs until the reset, and an immediate retry
109
+ fails. Do not retry.
110
+ 2. After the reset, reread the run directory and the ledger.
111
+ 3. Relaunch only the slices with no `<slice>.md`. Narrow each one if you can,
112
+ and give each an already-settled list built from the saved files.
113
+ 4. Never relaunch above the cap that hit the limit. If 4 in flight hit it,
114
+ relaunch at 3 or fewer.
115
+
116
+ **Any other error** (overloaded, or a rate limit with no reset time):
117
+
118
+ 1. Wait a minute or two, then relaunch that one slice once.
119
+ 2. If it fails again, review the slice yourself, or mark its ledger rows
120
+ `not reviewable (agent failed twice: <error>)`.
121
+
122
+ **Always:**
123
+
124
+ - Never relaunch a slice whose file exists.
125
+ - Never relaunch a whole wave.
126
+
127
+ ## If you were spawned as a subagent yourself
128
+
129
+ You are not the orchestrator. Do not fan out. Work the ledger yourself, and
130
+ return your report as your final message.
@@ -110,6 +110,12 @@ You have write access. Use it in proportion:
110
110
 
111
111
  - **Small and unambiguous** → fix it, commit, push, then re-review your own
112
112
  change as part of the diff. After pushing, wait for CI before approving.
113
+ **Locally**, a fix checks out the PR branch in a checkout other sessions may
114
+ share. Run `checkout acquire --card <slug>` before the checkout (`held` →
115
+ flag the fix instead of making it), `checkout acquire --card <slug> --branch
116
+ <pr-branch>` after it, and `checkout release --card <slug>` once the fix is
117
+ pushed and the tree is restored — protocol in
118
+ [conveyor-build's checkout-claim reference](../conveyor-build/references/checkout-claim.md).
113
119
  - **Larger, or a judgment call the author should make** → flag it in the
114
120
  verdict with the file, the line, what is wrong, and a suggested direction.
115
121
 
@@ -215,6 +215,13 @@ sync-spawned duplicate.
215
215
  origin <branch>` and `git merge-base --is-ancestor` before rebuilding
216
216
  anything locally — platform autosync may have already pushed for you, and a
217
217
  dirty tree may belong to a concurrent session.
218
+ - **Local sessions take turns on one checkout.** `conveyor-build`,
219
+ `conveyor-local-loop` and local `conveyor-review` fixes hold a checkout
220
+ claim, so a second session waits instead of switching branches underneath
221
+ the first. `conveyor-skills checkout status` shows who holds it (card,
222
+ branch, host, heartbeat age). `conveyor-skills checkout release --force` is
223
+ the human override for a wedged claim — an agent runs it only when the user
224
+ says so. Protocol: `conveyor-build/references/checkout-claim.md`.
218
225
 
219
226
  ## Improve This Skill
220
227