@staix/agent-hub 0.12.3 → 0.12.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -11,13 +11,108 @@ export function childEnv(source: NodeJS.ProcessEnv = process.env): NodeJS.Proces
11
11
  return env;
12
12
  }
13
13
 
14
+ /** One process: its identity is the pid with its start time (`lstart`, which exec keeps); group and command are evidence. */
15
+ export interface ProcRow {
16
+ pid: number;
17
+ ppid: number;
18
+ pgid: number;
19
+ started: string;
20
+ command: string;
21
+ }
22
+
23
+ /**
24
+ * `ps -axo pid=,ppid=,pgid=,stat=,lstart=,command=` output (unlimited width when not on a terminal), read in the C
25
+ * locale. A zombie is left out: it has exited, and only its parent's wait is missing.
26
+ */
27
+ export function parseProcessTable(text: string): ProcRow[] {
28
+ return text.split("\n").flatMap((line) => {
29
+ const m = /^\s*(\d+)\s+(\d+)\s+(\d+)\s+(\S+)\s+(\w{3} \w{3} [ \d]\d \d\d:\d\d:\d\d \d{4})\s+(.*)$/.exec(line);
30
+ return m && !m[4]!.startsWith("Z") ? [{ pid: Number(m[1]), ppid: Number(m[2]), pgid: Number(m[3]), started: m[5]!, command: m[6]! }] : [];
31
+ });
32
+ }
33
+
34
+ const PS = ["ps", "-axo", "pid=,ppid=,pgid=,stat=,lstart=,command="];
35
+ // `lstart` is local time: one zone for every reader, or a reader in another zone sees every identity as changed.
36
+ const PS_OPTIONS = { env: { ...process.env, LC_ALL: "C", TZ: "UTC" }, detached: true } as const;
37
+ /** The rows of one read, or undefined when they do not show the reader: a parse that found nothing is not a table. */
38
+ const own = (text: string) => { const rows = parseProcessTable(text); return rows.some((row) => row.pid === process.pid) ? rows : undefined; };
39
+
40
+ /**
41
+ * The process table, or undefined when it cannot be read or does not show this process: a parse that found nothing
42
+ * (a localized `lstart`, a changed `ps`) is not an empty table. Read with LC_ALL=C for that reason (and TZ=UTC, so
43
+ * start times compare across readers), by a `ps` in a
44
+ * process group of its own (a Ctrl-C to the caller's group would kill it), bounded, and tried twice.
45
+ */
46
+ export function processTable(): ProcRow[] | undefined {
47
+ for (let attempt = 0; attempt < 2; attempt++) {
48
+ try {
49
+ const r = Bun.spawnSync(PS, { ...PS_OPTIONS, stdout: "pipe", stderr: "pipe", timeout: 10_000 });
50
+ const rows = r.exitCode === 0 ? own(r.stdout.toString()) : undefined;
51
+ if (rows) return rows;
52
+ } catch {
53
+ // tried again below
54
+ }
55
+ }
56
+ return undefined;
57
+ }
58
+
59
+ /** `processTable`, without blocking the event loop: a hub stop reads the table many times while it keeps serving. */
60
+ export async function readProcessTable(): Promise<ProcRow[] | undefined> {
61
+ for (let attempt = 0; attempt < 2; attempt++) {
62
+ try {
63
+ const p = Bun.spawn(PS, { ...PS_OPTIONS, stdout: "pipe", stderr: "ignore" });
64
+ const timer = setTimeout(() => p.kill("SIGKILL"), 10_000);
65
+ const [text, code] = await Promise.all([new Response(p.stdout).text(), p.exited]).finally(() => clearTimeout(timer));
66
+ const rows = code === 0 ? own(text) : undefined;
67
+ if (rows) return rows;
68
+ } catch {
69
+ // tried again below
70
+ }
71
+ }
72
+ return undefined;
73
+ }
74
+
75
+ /** Every descendant of `pid` in `rows`, by parent links. */
76
+ export function descendantsOf(rows: ProcRow[], pid: number): ProcRow[] {
77
+ const out: ProcRow[] = [];
78
+ const parents = new Set([pid]);
79
+ for (let grew = true; grew; ) {
80
+ grew = false;
81
+ for (const row of rows) {
82
+ if (parents.has(row.ppid) && !parents.has(row.pid)) {
83
+ parents.add(row.pid);
84
+ out.push(row);
85
+ grew = true;
86
+ }
87
+ }
88
+ }
89
+ return out;
90
+ }
91
+
14
92
  /**
15
93
  * Stop a child owned by this hub and confirm that it exited. A timeout is an
16
94
  * incomplete shutdown, even after SIGKILL, because callers must not reuse a
17
95
  * port or claim ownership while the old process may still be alive.
96
+ *
97
+ * `group`: the child was spawned `detached` and leads its own process group; see `stopGroup`. `table` replaces the
98
+ * process table read (tests).
18
99
  */
19
- export async function stopOwnedProcess(proc: ChildProcess, termMs = 1_000, killMs = 2_000): Promise<void> {
20
- if (proc.exitCode !== null || proc.signalCode !== null || proc.pid === undefined) return;
100
+ export async function stopOwnedProcess(proc: ChildProcess, { termMs = 1_000, killMs = 2_000, group = false, table = readProcessTable }: { termMs?: number; killMs?: number; group?: boolean; table?: () => ProcRow[] | undefined | Promise<ProcRow[] | undefined> } = {}): Promise<void> {
101
+ if (proc.exitCode !== null || proc.signalCode !== null || proc.pid === undefined) {
102
+ // ponytail: a leader that exited before the stop (its launcher killed from outside) is not swept, as its pid may be
103
+ // reused once its group empties; its pipes are dropped so a survivor cannot keep the hub alive, and a group that
104
+ // still has members fails the stop instead of reading as done. Record the leader's start time at spawn to sweep it.
105
+ if (!group || proc.pid === undefined) return;
106
+ dropPipes(proc);
107
+ if (emptied.has(proc)) return; // its group was seen gone since: the id may be someone else's now
108
+ // A pid is not given out while a group with that id exists: a live process with the leader's pid means the group was
109
+ // emptied and the id is someone else's now. Members without it are what the leader left (zombies are not listed).
110
+ const rows = await table();
111
+ const members = rows ? (rows.some((r) => r.pid === proc.pid) ? [] : rows.filter((r) => r.pgid === proc.pid)) : undefined;
112
+ if (members ? members.length : !groupGone(proc.pid)) throw new Error(`owned child ${proc.pid} exited before the stop and its process group still has members${members ? ` (${members.map((r) => r.pid).join(", ")})` : ""}: not signalled; stop them to restart it`);
113
+ return;
114
+ }
115
+ if (group) return stopGroup(proc, proc.pid, termMs, killMs, table);
21
116
 
22
117
  const waitForExit = (timeoutMs: number): Promise<boolean> =>
23
118
  new Promise((resolve) => {
@@ -53,3 +148,129 @@ export async function stopOwnedProcess(proc: ChildProcess, termMs = 1_000, killM
53
148
  if (proc.exitCode !== null || proc.signalCode !== null || (await waitForExit(killMs))) return;
54
149
  throw new Error(`owned child ${proc.pid} shutdown incomplete after SIGKILL`);
55
150
  }
151
+
152
+ const dropPipes = (proc: ChildProcess) => {
153
+ proc.stdout?.destroy();
154
+ proc.stderr?.destroy();
155
+ };
156
+
157
+ const emptied = new WeakSet<ChildProcess>();
158
+ /**
159
+ * Follows the group of a child spawned `detached` after the child exits, until the group is gone: a later stop then
160
+ * never inspects the group id, which can be reused once the group is empty (issue #113).
161
+ */
162
+ export function trackGroup(proc: ChildProcess): void {
163
+ proc.once("exit", () => {
164
+ const pid = proc.pid;
165
+ if (pid === undefined) return;
166
+ const check = () => { if (groupGone(pid)) emptied.add(proc); else setTimeout(check, 1_000).unref(); };
167
+ check();
168
+ });
169
+ }
170
+
171
+ /** Whether no process is left in group `pgid` (macOS answers EPERM for a group of zombies). */
172
+ function groupGone(pgid: number): boolean {
173
+ try { process.kill(-pgid, 0); return false; } catch (error) {
174
+ const code = (error as NodeJS.ErrnoException).code;
175
+ return code === "ESRCH" || (code === "EPERM" && process.platform === "darwin");
176
+ }
177
+ }
178
+
179
+ /**
180
+ * Stops a child that leads its own process group, with everything it started (issue #113). A launcher that forwards
181
+ * signals to a native child (Codex's `codex.js`) otherwise dies alone at SIGKILL while its child, still at work, is
182
+ * re-parented to init with the launcher's pipes and keeps the hub alive; and the native child starts processes that
183
+ * lead groups of their own (Codex's MCP servers and tool commands), which no group signal reaches.
184
+ *
185
+ * SIGTERM to the group, a grace period in which what the tree starts is recorded while its parent still runs, then
186
+ * freeze, enumerate, kill: what is left is stopped (SIGSTOP) before the table is read again, because a stopped process
187
+ * starts nothing, so that read sees all of it; then SIGKILL. A snapshot taken while the tree runs misses what it starts
188
+ * next (three review rounds found such a gap). Done means the table shows none of it; without any table, the group's
189
+ * own answer (ESRCH, or EPERM on macOS for a group of zombies) is used, which cannot see groups below it.
190
+ */
191
+ async function stopGroup(proc: ChildProcess, pid: number, termMs: number, killMs: number, table: () => ProcRow[] | undefined | Promise<ProcRow[] | undefined>): Promise<void> {
192
+ const key = (r: { pid: number; started: string }) => `${r.pid}@${r.started}`;
193
+ const found = new Map<string, ProcRow>();
194
+ const exited = () => proc.exitCode !== null || proc.signalCode !== null;
195
+ let leader: ProcRow | undefined;
196
+ const live = (rows: ProcRow[], f: ProcRow) => rows.some((r) => r.pid === f.pid && r.started === f.started);
197
+ /** Reads the table and records what runs below the leader or anything recorded that still runs. */
198
+ const look = async (): Promise<ProcRow[] | undefined> => {
199
+ const rows = await table();
200
+ if (!rows) return undefined;
201
+ if (!leader && !exited()) leader = rows.find((r) => r.pid === pid); // never a pid taken over after the leader exited
202
+ const roots = [...(leader && live(rows, leader) ? [leader] : []), ...[...found.values()].filter((f) => live(rows, f))];
203
+ for (const root of roots) for (const r of descendantsOf(rows, root.pid)) found.set(key(r), r);
204
+ // A group is followed while it is known to be the same one: its leader or a recorded member still in it.
205
+ for (const id of [pid, ...[...found.values()].filter((f) => f.pgid === f.pid).map((f) => f.pid)]) {
206
+ // Current rows, never a recorded pgid: a member may have left the group since it was recorded.
207
+ const known = (id === pid && leader && live(rows, leader)) || rows.some((r) => r.pgid === id && r.pid !== id && found.has(key(r))) || rows.some((r) => r.pid === id && r.pgid === id && found.has(key(r)));
208
+ if (known) for (const r of rows) if (r.pgid === id && r.pid !== pid) found.set(key(r), r);
209
+ }
210
+ return rows;
211
+ };
212
+ const left = (rows: ProcRow[]) => [...found.values()].filter((f) => live(rows, f));
213
+ /**
214
+ * Running in the leader's group, or the group of a recorded process, without being recorded: after an empty moment the
215
+ * group id can be someone else's, so it is never signalled, but the stop is not done while it runs.
216
+ */
217
+ const unproven = (rows: ProcRow[]) => {
218
+ const ids = new Set([pid, ...[...found.values()].filter((f) => f.pgid === f.pid).map((f) => f.pid)]);
219
+ return rows.filter((r) => ids.has(r.pgid) && r.pid !== process.pid && !found.has(key(r)) && !(leader && r.pid === leader.pid && r.started === leader.started));
220
+ };
221
+ // Errors are not results here: the table read back decides. A sent signal says the target existed at that moment.
222
+ const signal = (target: number, sig: NodeJS.Signals) => { try { process.kill(target, sig); return true; } catch { return false; } };
223
+ const each = (rows: ProcRow[], sig: NodeJS.Signals) => rows.filter((r) => signal(r.pgid === r.pid ? -r.pid : r.pid, sig));
224
+ /** A read that takes longer than `ms` is given up: a frozen tree must not wait on a slow `ps` for its SIGKILL. */
225
+ const within = (ms: number) => Promise.race([look(), Bun.sleep(ms).then(() => undefined)]);
226
+ // What leads a group of its own (an MCP server, a tool command) gets a SIGTERM of its own and the same grace period:
227
+ // a git process stopped by SIGKILL leaves its index lock behind.
228
+ const termed = new Set<string>();
229
+ const term = (rows: ProcRow[] | undefined) => {
230
+ for (const r of rows ? left(rows) : []) if (r.pgid === r.pid && !termed.has(key(r))) { termed.add(key(r)); signal(-r.pid, "SIGTERM"); }
231
+ };
232
+ try {
233
+ const first = await look();
234
+ if (!exited()) signal(-pid, "SIGTERM"); // a reaped leader's group id is not proven any more
235
+ term(first);
236
+ // The grace period, for the leader and for what it started: what they start meanwhile is recorded while its parent
237
+ // still runs.
238
+ for (const end = Date.now() + termMs; Date.now() < end; ) {
239
+ await Bun.sleep(100);
240
+ const rows = await look();
241
+ term(rows);
242
+ if (exited() && rows && !left(rows).length) break;
243
+ }
244
+ // Freeze, enumerate, kill: a stopped process starts nothing, so a read after the freeze sees all of it.
245
+ for (const end = Date.now() + killMs; ; ) {
246
+ const rows = await look();
247
+ const rest = rows ? left(rows) : undefined;
248
+ const strangers = rows ? unproven(rows) : [];
249
+ if (exited() && !strangers.length && (rest ? !rest.length : !found.size && groupGone(pid))) return;
250
+ if (Date.now() >= end) {
251
+ throw new Error(`owned child ${pid}: ${rest ? `${rest.length + strangers.length} process(es) of its group or below it still running${strangers.length ? `, ${strangers.length} not proven its own and left alone` : ""}` : "the process table cannot be read to confirm what it started is gone"}`);
252
+ }
253
+ // What is stopped is killed or continued, never left frozen: the group by whether its STOP was sent (its id stays
254
+ // reserved while a member lives), each process by whether the read after the freeze still shows it. What a read
255
+ // after the freeze shows for the first time (started between the read and the STOPs, in a group of its own) is
256
+ // frozen too, and read again, before anything is killed.
257
+ const groupStopped = !exited() && signal(-pid, "SIGSTOP");
258
+ const stopped = new Map<string, ProcRow>();
259
+ let frozen: ProcRow[] | undefined;
260
+ for (let todo = rest ?? [], round = 0; ; round++) {
261
+ for (const r of each(todo, "SIGSTOP")) stopped.set(key(r), r);
262
+ frozen = await within(1_000);
263
+ todo = frozen ? left(frozen).filter((r) => !stopped.has(key(r))) : [];
264
+ if (!todo.length || round >= 4) break;
265
+ }
266
+ // Without a read after the freeze, only what the STOP reached is killed: a stopped process keeps its pid.
267
+ const kill = frozen ? left(frozen) : [...stopped.values()];
268
+ each(kill, "SIGKILL");
269
+ if (groupStopped) signal(-pid, "SIGKILL");
270
+ for (const r of stopped.values()) if (!kill.some((k) => k.pid === r.pid && k.started === r.started)) signal(r.pgid === r.pid ? -r.pid : r.pid, "SIGCONT");
271
+ await Bun.sleep(50);
272
+ }
273
+ } finally {
274
+ dropPipes(proc); // a process that left the tree between two reads may still hold them: it must not keep the hub alive
275
+ }
276
+ }
@@ -0,0 +1,308 @@
1
+ import type { Task } from "./board.ts";
2
+ import type { PeerId } from "./envelope.ts";
3
+
4
+ /**
5
+ * Turn-free cohorts (issue #107): the owners of overlapping tasks, frozen into one group from the moment the overlap is
6
+ * found until each of them has stopped after its task closed. Whether a cohort is silent is decided when it is formed
7
+ * (turn-free on, no PII, every owner's context path verified) and never switched on later; it is lifted when an owner
8
+ * that cannot receive facts joins, a path is lost, or a PII task opens. Membership does not end when the board closes a
9
+ * task: a member stays in until its native turn has ended after that, so a late final answer is still the cohort's.
10
+ * That settlement is recorded when it happens and never undone: a settled member's later turns are new work.
11
+ *
12
+ * Completion is two-step. Each member's done is an intent. The member whose intent completes the set is selected, in
13
+ * one synchronous step, to integrate: it is asked to check its work against the others before its done is recorded,
14
+ * and its next done is accepted only for the same target (owner generation, cohort revision, files) once every other
15
+ * member has settled. Anything that changes the target asks again, a bounded number of times; past that the outcome is
16
+ * recorded as unresolved, never as integrated.
17
+ */
18
+
19
+ /** Integration requests for one cohort revision before the outcome is recorded as unresolved. */
20
+ export const MAX_REQUESTS = 3;
21
+ /** A done this soon after an integration request is a retry of a lost answer, not a check of the work. */
22
+ export const RETRY_MS = 2000;
23
+
24
+ export interface Member {
25
+ task: number;
26
+ owner: PeerId;
27
+ /** Changes whenever the task changes hands. */
28
+ gen: number;
29
+ /** When its owner was handed the task: its writes count for the integration target from here until it settles. */
30
+ since: number;
31
+ /** When the task left the open states for its owner. */
32
+ closedAt?: number;
33
+ /** When its owner's native turn first ended after that (or the task closed while the owner was idle). */
34
+ settledAt?: number;
35
+ /** When its owner's native turn first ended after its current intent: its writes for the task have stopped. */
36
+ stoppedAt?: number;
37
+ }
38
+
39
+ export interface Integration {
40
+ task: number;
41
+ owner: PeerId;
42
+ gen: number;
43
+ revision: number;
44
+ /** The files as they were at the last request: the next done is accepted only for this target. */
45
+ tree: string;
46
+ requests: number;
47
+ /** When the last request went out: a done right after it is a retry, not a confirmation. */
48
+ at: number;
49
+ /** The fact offer that went out with the last request: the next done acknowledges it. */
50
+ offer?: string;
51
+ confirmed?: boolean;
52
+ /** The outcome was recorded as unresolved: this revision asks nothing more, and the member's check counts as usual. */
53
+ closed?: boolean;
54
+ }
55
+
56
+ export interface Cohort {
57
+ id: number;
58
+ /** Bumps on every change of membership or owner, when an intent is withdrawn, and when the cohort is lifted. */
59
+ revision: number;
60
+ members: Map<number, Member>;
61
+ silent: boolean;
62
+ intents: Map<number, { gen: number; at: number }>;
63
+ integration?: Integration;
64
+ /** Members whose completed-change notice the silence withheld: what replaces it if no integration runs. */
65
+ held: Set<number>;
66
+ }
67
+
68
+ export interface CohortDeps {
69
+ /** Whether these owners may work without messages: turn-free on, no PII, every owner's context path verified. */
70
+ silence: (owners: PeerId[]) => boolean;
71
+ /**
72
+ * Whether `peer` is between native turns now (Codex not busy; Claude's last tool call started before its last Stop).
73
+ * Evidence is a native turn end, never a task approval or a delivery acknowledgement.
74
+ */
75
+ idle: (peer: PeerId) => boolean;
76
+ }
77
+
78
+ export type Completion =
79
+ | { action: "proceed"; integrated?: boolean }
80
+ | { action: "request"; why: string; requests: number; cohort: Cohort }
81
+ | { action: "unresolved"; why: string; cohort: Cohort };
82
+
83
+ export class Cohorts {
84
+ private readonly live: Cohort[] = [];
85
+ private next = 1;
86
+
87
+ constructor(private readonly d: CohortDeps) {}
88
+
89
+ /** The live cohort `task` is in. A cohort is over the moment its members have all finished, whoever asks. */
90
+ of(task: number): Cohort | undefined {
91
+ this.gc();
92
+ return this.live.find((c) => c.members.has(task));
93
+ }
94
+
95
+ list(): Cohort[] {
96
+ this.gc();
97
+ return [...this.live];
98
+ }
99
+
100
+ /**
101
+ * `task` overlaps `others` (open tasks of other owners): from now on they are one cohort. Merges cohorts the tasks were
102
+ * in, records owner changes (each bumps the revision and voids that member's intent), and lifts a silent cohort that
103
+ * an owner without verified facts joins. Returns what happened, for the notices.
104
+ */
105
+ join(task: Task, others: Task[], gen: (t: Task) => number, handed: (t: Task) => number = () => Date.now()): { cohort: Cohort; formed: boolean; lifted: boolean } | undefined {
106
+ this.gc();
107
+ const all = [task, ...others].filter((t) => t.owner);
108
+ const found = [...new Set(all.map((t) => this.of(t.id)).filter((c): c is Cohort => !!c))];
109
+ if (!found.length && all.length < 2) return undefined;
110
+ const wasSilent = found.some((c) => c.silent);
111
+ let cohort = found[0];
112
+ const formed = !cohort;
113
+ if (!cohort) {
114
+ cohort = { id: this.next++, revision: 0, members: new Map(), silent: false, intents: new Map(), held: new Set() };
115
+ this.live.push(cohort);
116
+ }
117
+ let changed = formed;
118
+ for (const other of found.slice(1)) {
119
+ for (const [id, m] of other.members) cohort.members.set(id, m);
120
+ for (const [id, i] of other.intents) cohort.intents.set(id, i);
121
+ for (const id of other.held) cohort.held.add(id);
122
+ cohort.silent &&= other.silent;
123
+ this.live.splice(this.live.indexOf(other), 1);
124
+ changed = true;
125
+ }
126
+ for (const t of all) {
127
+ const m = cohort.members.get(t.id);
128
+ const g = gen(t);
129
+ if (m && m.owner === t.owner && m.gen === g) continue;
130
+ cohort.members.set(t.id, { task: t.id, owner: t.owner!, gen: g, since: handed(t) });
131
+ cohort.intents.delete(t.id);
132
+ changed = true;
133
+ }
134
+ if (changed) cohort.revision++;
135
+ const owners = [...new Set([...cohort.members.values()].map((m) => m.owner))];
136
+ if (formed) cohort.silent = this.d.silence(owners);
137
+ else if (cohort.silent && changed && !this.d.silence(owners)) cohort.silent = false;
138
+ // Lifted: a silent cohort (or a silent one merged into this) is no longer silent, and its members must hear it.
139
+ return { cohort, formed, lifted: wasSilent && !cohort.silent };
140
+ }
141
+
142
+ /** An owner lost its context path: every silent cohort it is in speaks again. */
143
+ lift(peer: PeerId): Cohort[] {
144
+ return this.liftWhere((c) => [...c.members.values()].some((m) => m.owner === peer));
145
+ }
146
+
147
+ /** A PII task opened: every silent cohort speaks again, at once (issue #108). */
148
+ liftAll(): Cohort[] {
149
+ return this.liftWhere(() => true);
150
+ }
151
+
152
+ private liftWhere(pick: (c: Cohort) => boolean): Cohort[] {
153
+ this.gc(); // a finished cohort is not lifted: a peer leaving after the work (a benchmark's teardown) changes nothing
154
+ const out = this.live.filter((c) => c.silent && pick(c));
155
+ for (const c of out) {
156
+ c.silent = false;
157
+ c.revision++;
158
+ }
159
+ return out;
160
+ }
161
+
162
+ /**
163
+ * The silent cohort that makes a message from `from` to `to` cohort coordination: `from` is a member that has not
164
+ * settled since its task closed, and `to` is a member too. A settled member works on something else now.
165
+ */
166
+ silenced(from: PeerId, to: PeerId): Cohort | undefined {
167
+ this.gc();
168
+ return this.live.find((c) => {
169
+ if (!c.silent) return false;
170
+ const members = [...c.members.values()];
171
+ return members.some((m) => m.owner === from && m.settledAt === undefined) && members.some((m) => m.owner === to);
172
+ });
173
+ }
174
+
175
+ /** The task left the open states for its owner (done, approved, in review); an idle owner settles at once. */
176
+ closed(task: number, at = Date.now()): void {
177
+ const m = this.of(task)?.members.get(task);
178
+ if (!m) return;
179
+ m.closedAt ??= at;
180
+ if (m.settledAt === undefined && this.d.idle(m.owner)) m.settledAt = at;
181
+ }
182
+
183
+ /** `peer`'s native turn ended: its members whose tasks closed before settle, and those with an intent stop, for good. */
184
+ turnEnded(peer: PeerId, at = Date.now()): void {
185
+ for (const c of this.live) {
186
+ for (const m of c.members.values()) {
187
+ if (m.owner !== peer) continue;
188
+ if (m.closedAt !== undefined && m.closedAt <= at && m.settledAt === undefined) m.settledAt = at;
189
+ const intent = c.intents.get(m.task);
190
+ if (intent && intent.gen === m.gen && intent.at <= at && m.stoppedAt === undefined) m.stoppedAt = at;
191
+ }
192
+ }
193
+ }
194
+
195
+ /**
196
+ * The task is open again for its owner (a failed check, changes requested, a reopen): its intent is void, and while
197
+ * its cohort is live it is that cohort's work again. A cohort that is over stays over.
198
+ */
199
+ withdraw(task: number): void {
200
+ const c = this.of(task);
201
+ if (!c) return;
202
+ const m = c.members.get(task);
203
+ if (m) {
204
+ delete m.closedAt;
205
+ delete m.settledAt;
206
+ delete m.stoppedAt;
207
+ }
208
+ c.intents.delete(task);
209
+ c.revision++;
210
+ }
211
+
212
+ /**
213
+ * A done of `task` that selects nobody and asks nothing: the console finishing a member (issue #107). It still counts
214
+ * as that member's intent, so the member whose done completes the set integrates.
215
+ */
216
+ intent(c: Cohort, task: Task, gen: number): void {
217
+ if (c.intents.has(task.id) && c.intents.get(task.id)!.gen === gen) return;
218
+ c.intents.set(task.id, { gen, at: Date.now() });
219
+ const m = c.members.get(task.id);
220
+ if (!m) return;
221
+ // An owner between turns (the console finished the task, say) has stopped already; one in a turn stops when it ends.
222
+ if (this.d.idle(m.owner)) m.stoppedAt = Date.now();
223
+ else delete m.stoppedAt;
224
+ }
225
+
226
+ /** Whether this done of the integrating member is a retry of a lost answer: then nothing counts or is acknowledged. */
227
+ isRetry(c: Cohort, task: Task, gen: number): boolean {
228
+ const ig = c.integration;
229
+ return !!ig && ig.task === task.id && ig.gen === gen && ig.owner === task.owner && ig.revision === c.revision && !ig.confirmed && Date.now() - ig.at < RETRY_MS;
230
+ }
231
+
232
+ /**
233
+ * `task`'s owner called done in silent cohort `c`. Records its intent; then either it proceeds (others are still at
234
+ * work, another member integrates this revision, or its integration is confirmed), or it is asked to integrate (first
235
+ * request, or the target moved), or, past MAX_REQUESTS, the outcome is unresolved.
236
+ */
237
+ completion(c: Cohort, task: Task, now: { gen: number; tree: string; factsCurrent: boolean; handed?: number }): Completion {
238
+ const m = c.members.get(task.id);
239
+ if (!m || m.owner !== task.owner || m.gen !== now.gen) {
240
+ c.members.set(task.id, { task: task.id, owner: task.owner!, gen: now.gen, since: now.handed ?? Date.now() });
241
+ c.revision++;
242
+ }
243
+ this.intent(c, task, now.gen);
244
+ const missing = [...c.members.values()].filter((x) => c.intents.get(x.task)?.gen !== x.gen);
245
+ if (missing.length) return { action: "proceed" };
246
+ const ig = c.integration;
247
+ if (ig && ig.revision === c.revision && ig.task !== task.id) return { action: "proceed" }; // another member integrates
248
+ if (!ig || ig.revision !== c.revision || ig.task !== task.id || ig.gen !== now.gen || ig.owner !== task.owner) {
249
+ c.integration = { task: task.id, owner: task.owner!, gen: now.gen, revision: c.revision, tree: now.tree, requests: 1, at: Date.now() };
250
+ return { action: "request", why: "", requests: 1, cohort: c };
251
+ }
252
+ if (ig.closed) return { action: "proceed" }; // recorded as unresolved: nothing more is asked of this revision
253
+ // Its next done: accepted only for the same target, once every other member has settled. A later turn of a settled
254
+ // member that touches these files moves the target, which the file hash catches.
255
+ // Other owners only: the integrating owner's own other tasks in the cohort stop with this very turn.
256
+ const running = [...c.members.values()].filter((x) => x.owner !== task.owner && x.stoppedAt === undefined).map((x) => x.owner);
257
+ let why = "";
258
+ if (running.length) why = `${[...new Set(running)].join(", ")} has not stopped since its done, so its changes may not be final`;
259
+ else if (now.tree !== ig.tree) why = "the files changed since the last request";
260
+ else if (!now.factsCurrent) why = "there are changes you have not been shown yet";
261
+ if (!why) {
262
+ ig.confirmed = true;
263
+ return { action: "proceed", integrated: true };
264
+ }
265
+ if (ig.requests >= MAX_REQUESTS) {
266
+ ig.closed = true;
267
+ return { action: "unresolved", why, cohort: c };
268
+ }
269
+ ig.requests++;
270
+ ig.tree = now.tree;
271
+ ig.at = Date.now();
272
+ delete ig.offer;
273
+ delete ig.confirmed; // asked again: a done right after this is a retry, and only a later one can confirm
274
+ return { action: "request", why, requests: ig.requests, cohort: c };
275
+ }
276
+
277
+ /**
278
+ * Whether a check that passed for `task` counts: always, unless `task` is the confirmed integrating member of its
279
+ * cohort's current revision, whose check counts only for the target it confirmed. A member that stopped being the
280
+ * last to finish (the revision moved) is an earlier finisher again, and its own check counts.
281
+ */
282
+ holds(task: number, gen: number, tree: string): boolean {
283
+ const c = this.of(task);
284
+ const ig = c?.integration;
285
+ if (!c || !c.silent || !ig || ig.task !== task || ig.revision !== c.revision || ig.closed) return true;
286
+ return !!ig.confirmed && ig.gen === gen && ig.tree === tree;
287
+ }
288
+
289
+ /** The task lost its owner (released, unassigned): it leaves the cohort, and the revision moves on. */
290
+ leave(task: number): void {
291
+ const c = this.of(task);
292
+ if (!c) return;
293
+ c.members.delete(task);
294
+ c.intents.delete(task);
295
+ c.revision++;
296
+ }
297
+
298
+ /**
299
+ * A cohort is over once every member closed its task and settled (silent), or closed it (not silent: nothing is held
300
+ * back, so nothing waits for a turn end).
301
+ */
302
+ private gc(): void {
303
+ for (const c of [...this.live]) {
304
+ const members = [...c.members.values()];
305
+ if (members.every((m) => (c.silent ? m.settledAt !== undefined : m.closedAt !== undefined))) this.live.splice(this.live.indexOf(c), 1);
306
+ }
307
+ }
308
+ }
@@ -6,10 +6,10 @@ export function stateDirFor(cwd: string): string {
6
6
  return projectContext(cwd).stateDir;
7
7
  }
8
8
 
9
- /** Control WS wire version. 2 = `deliver` carries `envs` (digests); 3 = `tools` role and task messages; 4 = budget messages and `hub_checkpoint`; 5 = `ask`; 6 = console-only `ui` session bootstrap; 8 = controlled recovery; 9 = Pi bridge metadata; 10 = durable delivery receipts; 11 = queue hold diagnostics; 12 = generation-bound channel settlement and execution budgets. The plugin is installed apart from the daemon, so they can drift. */
10
- export const PROTOCOL = 12;
9
+ /** Control WS wire version. 2 = `deliver` carries `envs` (digests); 3 = `tools` role and task messages; 4 = budget messages and `hub_checkpoint`; 5 = `ask`; 6 = console-only `ui` session bootstrap; 8 = controlled recovery; 9 = Pi bridge metadata; 10 = durable delivery receipts; 11 = queue hold diagnostics; 12 = generation-bound channel settlement and execution budgets; 13 = turn-free facts (`facts` with tool, session and transcript binding, `silenced`) and per-recipient send results (issue #108). The plugin is installed apart from the daemon, so they can drift. */
10
+ export const PROTOCOL = 13;
11
11
  /** Protocols a current coordinator may authenticate while upgrading a running source. */
12
- export const RECOVERY_SOURCE_PROTOCOLS = [9, 10, 11, PROTOCOL] as const;
12
+ export const RECOVERY_SOURCE_PROTOCOLS = [9, 10, 11, 12, PROTOCOL] as const;
13
13
 
14
14
  export interface Hello {
15
15
  /** `tools`: acts for `peer` (task tools, hub_send) without being a delivery target: the MCP server Kimi and Codex run. */