@staix/agent-hub 0.12.3 → 0.12.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -0
- package/docs/cooperbench.md +35 -2
- package/docs/events.md +9 -0
- package/docs/operations.md +173 -5
- package/docs/specs/2026-09-19-agent-hub-design.md +31 -1
- package/docs/verification/2026-10-02-0.12.4.md +82 -0
- package/docs/verification/2026-10-03-0.12.5.md +46 -0
- package/package.json +1 -1
- package/plugins/agent-hub/.claude-plugin/plugin.json +1 -1
- package/plugins/agent-hub/server.js +5 -4
- package/src/adapters/claude-channel.ts +2 -1
- package/src/adapters/codex-appserver.ts +51 -6
- package/src/cli/facts-hook.ts +40 -0
- package/src/cli/launch.ts +27 -3
- package/src/cli/main.ts +31 -2
- package/src/cli/upgrade-runtime.ts +1 -1
- package/src/hub/board.ts +4 -2
- package/src/hub/bus.ts +75 -3
- package/src/hub/child-process.ts +223 -2
- package/src/hub/cohorts.ts +308 -0
- package/src/hub/control-client.ts +3 -3
- package/src/hub/daemon.ts +389 -15
- package/src/hub/events.ts +9 -0
- package/src/hub/facts.ts +853 -0
- package/src/hub/report.ts +3 -0
- package/src/hub/routing.ts +94 -0
- package/src/hub/tasks.ts +412 -20
- package/templates/AGENTS.block.md +1 -1
package/src/hub/child-process.ts
CHANGED
|
@@ -11,13 +11,108 @@ export function childEnv(source: NodeJS.ProcessEnv = process.env): NodeJS.Proces
|
|
|
11
11
|
return env;
|
|
12
12
|
}
|
|
13
13
|
|
|
14
|
+
/** One process: its identity is the pid with its start time (`lstart`, which exec keeps); group and command are evidence. */
|
|
15
|
+
export interface ProcRow {
|
|
16
|
+
pid: number;
|
|
17
|
+
ppid: number;
|
|
18
|
+
pgid: number;
|
|
19
|
+
started: string;
|
|
20
|
+
command: string;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* `ps -axo pid=,ppid=,pgid=,stat=,lstart=,command=` output (unlimited width when not on a terminal), read in the C
|
|
25
|
+
* locale. A zombie is left out: it has exited, and only its parent's wait is missing.
|
|
26
|
+
*/
|
|
27
|
+
export function parseProcessTable(text: string): ProcRow[] {
|
|
28
|
+
return text.split("\n").flatMap((line) => {
|
|
29
|
+
const m = /^\s*(\d+)\s+(\d+)\s+(\d+)\s+(\S+)\s+(\w{3} \w{3} [ \d]\d \d\d:\d\d:\d\d \d{4})\s+(.*)$/.exec(line);
|
|
30
|
+
return m && !m[4]!.startsWith("Z") ? [{ pid: Number(m[1]), ppid: Number(m[2]), pgid: Number(m[3]), started: m[5]!, command: m[6]! }] : [];
|
|
31
|
+
});
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
const PS = ["ps", "-axo", "pid=,ppid=,pgid=,stat=,lstart=,command="];
|
|
35
|
+
// `lstart` is local time: one zone for every reader, or a reader in another zone sees every identity as changed.
|
|
36
|
+
const PS_OPTIONS = { env: { ...process.env, LC_ALL: "C", TZ: "UTC" }, detached: true } as const;
|
|
37
|
+
/** The rows of one read, or undefined when they do not show the reader: a parse that found nothing is not a table. */
|
|
38
|
+
const own = (text: string) => { const rows = parseProcessTable(text); return rows.some((row) => row.pid === process.pid) ? rows : undefined; };
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* The process table, or undefined when it cannot be read or does not show this process: a parse that found nothing
|
|
42
|
+
* (a localized `lstart`, a changed `ps`) is not an empty table. Read with LC_ALL=C for that reason (and TZ=UTC, so
|
|
43
|
+
* start times compare across readers), by a `ps` in a
|
|
44
|
+
* process group of its own (a Ctrl-C to the caller's group would kill it), bounded, and tried twice.
|
|
45
|
+
*/
|
|
46
|
+
export function processTable(): ProcRow[] | undefined {
|
|
47
|
+
for (let attempt = 0; attempt < 2; attempt++) {
|
|
48
|
+
try {
|
|
49
|
+
const r = Bun.spawnSync(PS, { ...PS_OPTIONS, stdout: "pipe", stderr: "pipe", timeout: 10_000 });
|
|
50
|
+
const rows = r.exitCode === 0 ? own(r.stdout.toString()) : undefined;
|
|
51
|
+
if (rows) return rows;
|
|
52
|
+
} catch {
|
|
53
|
+
// tried again below
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
return undefined;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** `processTable`, without blocking the event loop: a hub stop reads the table many times while it keeps serving. */
|
|
60
|
+
export async function readProcessTable(): Promise<ProcRow[] | undefined> {
|
|
61
|
+
for (let attempt = 0; attempt < 2; attempt++) {
|
|
62
|
+
try {
|
|
63
|
+
const p = Bun.spawn(PS, { ...PS_OPTIONS, stdout: "pipe", stderr: "ignore" });
|
|
64
|
+
const timer = setTimeout(() => p.kill("SIGKILL"), 10_000);
|
|
65
|
+
const [text, code] = await Promise.all([new Response(p.stdout).text(), p.exited]).finally(() => clearTimeout(timer));
|
|
66
|
+
const rows = code === 0 ? own(text) : undefined;
|
|
67
|
+
if (rows) return rows;
|
|
68
|
+
} catch {
|
|
69
|
+
// tried again below
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
return undefined;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** Every descendant of `pid` in `rows`, by parent links. */
|
|
76
|
+
export function descendantsOf(rows: ProcRow[], pid: number): ProcRow[] {
|
|
77
|
+
const out: ProcRow[] = [];
|
|
78
|
+
const parents = new Set([pid]);
|
|
79
|
+
for (let grew = true; grew; ) {
|
|
80
|
+
grew = false;
|
|
81
|
+
for (const row of rows) {
|
|
82
|
+
if (parents.has(row.ppid) && !parents.has(row.pid)) {
|
|
83
|
+
parents.add(row.pid);
|
|
84
|
+
out.push(row);
|
|
85
|
+
grew = true;
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
return out;
|
|
90
|
+
}
|
|
91
|
+
|
|
14
92
|
/**
|
|
15
93
|
* Stop a child owned by this hub and confirm that it exited. A timeout is an
|
|
16
94
|
* incomplete shutdown, even after SIGKILL, because callers must not reuse a
|
|
17
95
|
* port or claim ownership while the old process may still be alive.
|
|
96
|
+
*
|
|
97
|
+
* `group`: the child was spawned `detached` and leads its own process group; see `stopGroup`. `table` replaces the
|
|
98
|
+
* process table read (tests).
|
|
18
99
|
*/
|
|
19
|
-
export async function stopOwnedProcess(proc: ChildProcess, termMs = 1_000, killMs = 2_000): Promise<void> {
|
|
20
|
-
if (proc.exitCode !== null || proc.signalCode !== null || proc.pid === undefined)
|
|
100
|
+
export async function stopOwnedProcess(proc: ChildProcess, { termMs = 1_000, killMs = 2_000, group = false, table = readProcessTable }: { termMs?: number; killMs?: number; group?: boolean; table?: () => ProcRow[] | undefined | Promise<ProcRow[] | undefined> } = {}): Promise<void> {
|
|
101
|
+
if (proc.exitCode !== null || proc.signalCode !== null || proc.pid === undefined) {
|
|
102
|
+
// ponytail: a leader that exited before the stop (its launcher killed from outside) is not swept, as its pid may be
|
|
103
|
+
// reused once its group empties; its pipes are dropped so a survivor cannot keep the hub alive, and a group that
|
|
104
|
+
// still has members fails the stop instead of reading as done. Record the leader's start time at spawn to sweep it.
|
|
105
|
+
if (!group || proc.pid === undefined) return;
|
|
106
|
+
dropPipes(proc);
|
|
107
|
+
if (emptied.has(proc)) return; // its group was seen gone since: the id may be someone else's now
|
|
108
|
+
// A pid is not given out while a group with that id exists: a live process with the leader's pid means the group was
|
|
109
|
+
// emptied and the id is someone else's now. Members without it are what the leader left (zombies are not listed).
|
|
110
|
+
const rows = await table();
|
|
111
|
+
const members = rows ? (rows.some((r) => r.pid === proc.pid) ? [] : rows.filter((r) => r.pgid === proc.pid)) : undefined;
|
|
112
|
+
if (members ? members.length : !groupGone(proc.pid)) throw new Error(`owned child ${proc.pid} exited before the stop and its process group still has members${members ? ` (${members.map((r) => r.pid).join(", ")})` : ""}: not signalled; stop them to restart it`);
|
|
113
|
+
return;
|
|
114
|
+
}
|
|
115
|
+
if (group) return stopGroup(proc, proc.pid, termMs, killMs, table);
|
|
21
116
|
|
|
22
117
|
const waitForExit = (timeoutMs: number): Promise<boolean> =>
|
|
23
118
|
new Promise((resolve) => {
|
|
@@ -53,3 +148,129 @@ export async function stopOwnedProcess(proc: ChildProcess, termMs = 1_000, killM
|
|
|
53
148
|
if (proc.exitCode !== null || proc.signalCode !== null || (await waitForExit(killMs))) return;
|
|
54
149
|
throw new Error(`owned child ${proc.pid} shutdown incomplete after SIGKILL`);
|
|
55
150
|
}
|
|
151
|
+
|
|
152
|
+
const dropPipes = (proc: ChildProcess) => {
|
|
153
|
+
proc.stdout?.destroy();
|
|
154
|
+
proc.stderr?.destroy();
|
|
155
|
+
};
|
|
156
|
+
|
|
157
|
+
const emptied = new WeakSet<ChildProcess>();
|
|
158
|
+
/**
|
|
159
|
+
* Follows the group of a child spawned `detached` after the child exits, until the group is gone: a later stop then
|
|
160
|
+
* never inspects the group id, which can be reused once the group is empty (issue #113).
|
|
161
|
+
*/
|
|
162
|
+
export function trackGroup(proc: ChildProcess): void {
|
|
163
|
+
proc.once("exit", () => {
|
|
164
|
+
const pid = proc.pid;
|
|
165
|
+
if (pid === undefined) return;
|
|
166
|
+
const check = () => { if (groupGone(pid)) emptied.add(proc); else setTimeout(check, 1_000).unref(); };
|
|
167
|
+
check();
|
|
168
|
+
});
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
/** Whether no process is left in group `pgid` (macOS answers EPERM for a group of zombies). */
|
|
172
|
+
function groupGone(pgid: number): boolean {
|
|
173
|
+
try { process.kill(-pgid, 0); return false; } catch (error) {
|
|
174
|
+
const code = (error as NodeJS.ErrnoException).code;
|
|
175
|
+
return code === "ESRCH" || (code === "EPERM" && process.platform === "darwin");
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* Stops a child that leads its own process group, with everything it started (issue #113). A launcher that forwards
|
|
181
|
+
* signals to a native child (Codex's `codex.js`) otherwise dies alone at SIGKILL while its child, still at work, is
|
|
182
|
+
* re-parented to init with the launcher's pipes and keeps the hub alive; and the native child starts processes that
|
|
183
|
+
* lead groups of their own (Codex's MCP servers and tool commands), which no group signal reaches.
|
|
184
|
+
*
|
|
185
|
+
* SIGTERM to the group, a grace period in which what the tree starts is recorded while its parent still runs, then
|
|
186
|
+
* freeze, enumerate, kill: what is left is stopped (SIGSTOP) before the table is read again, because a stopped process
|
|
187
|
+
* starts nothing, so that read sees all of it; then SIGKILL. A snapshot taken while the tree runs misses what it starts
|
|
188
|
+
* next (three review rounds found such a gap). Done means the table shows none of it; without any table, the group's
|
|
189
|
+
* own answer (ESRCH, or EPERM on macOS for a group of zombies) is used, which cannot see groups below it.
|
|
190
|
+
*/
|
|
191
|
+
async function stopGroup(proc: ChildProcess, pid: number, termMs: number, killMs: number, table: () => ProcRow[] | undefined | Promise<ProcRow[] | undefined>): Promise<void> {
|
|
192
|
+
const key = (r: { pid: number; started: string }) => `${r.pid}@${r.started}`;
|
|
193
|
+
const found = new Map<string, ProcRow>();
|
|
194
|
+
const exited = () => proc.exitCode !== null || proc.signalCode !== null;
|
|
195
|
+
let leader: ProcRow | undefined;
|
|
196
|
+
const live = (rows: ProcRow[], f: ProcRow) => rows.some((r) => r.pid === f.pid && r.started === f.started);
|
|
197
|
+
/** Reads the table and records what runs below the leader or anything recorded that still runs. */
|
|
198
|
+
const look = async (): Promise<ProcRow[] | undefined> => {
|
|
199
|
+
const rows = await table();
|
|
200
|
+
if (!rows) return undefined;
|
|
201
|
+
if (!leader && !exited()) leader = rows.find((r) => r.pid === pid); // never a pid taken over after the leader exited
|
|
202
|
+
const roots = [...(leader && live(rows, leader) ? [leader] : []), ...[...found.values()].filter((f) => live(rows, f))];
|
|
203
|
+
for (const root of roots) for (const r of descendantsOf(rows, root.pid)) found.set(key(r), r);
|
|
204
|
+
// A group is followed while it is known to be the same one: its leader or a recorded member still in it.
|
|
205
|
+
for (const id of [pid, ...[...found.values()].filter((f) => f.pgid === f.pid).map((f) => f.pid)]) {
|
|
206
|
+
// Current rows, never a recorded pgid: a member may have left the group since it was recorded.
|
|
207
|
+
const known = (id === pid && leader && live(rows, leader)) || rows.some((r) => r.pgid === id && r.pid !== id && found.has(key(r))) || rows.some((r) => r.pid === id && r.pgid === id && found.has(key(r)));
|
|
208
|
+
if (known) for (const r of rows) if (r.pgid === id && r.pid !== pid) found.set(key(r), r);
|
|
209
|
+
}
|
|
210
|
+
return rows;
|
|
211
|
+
};
|
|
212
|
+
const left = (rows: ProcRow[]) => [...found.values()].filter((f) => live(rows, f));
|
|
213
|
+
/**
|
|
214
|
+
* Running in the leader's group, or the group of a recorded process, without being recorded: after an empty moment the
|
|
215
|
+
* group id can be someone else's, so it is never signalled, but the stop is not done while it runs.
|
|
216
|
+
*/
|
|
217
|
+
const unproven = (rows: ProcRow[]) => {
|
|
218
|
+
const ids = new Set([pid, ...[...found.values()].filter((f) => f.pgid === f.pid).map((f) => f.pid)]);
|
|
219
|
+
return rows.filter((r) => ids.has(r.pgid) && r.pid !== process.pid && !found.has(key(r)) && !(leader && r.pid === leader.pid && r.started === leader.started));
|
|
220
|
+
};
|
|
221
|
+
// Errors are not results here: the table read back decides. A sent signal says the target existed at that moment.
|
|
222
|
+
const signal = (target: number, sig: NodeJS.Signals) => { try { process.kill(target, sig); return true; } catch { return false; } };
|
|
223
|
+
const each = (rows: ProcRow[], sig: NodeJS.Signals) => rows.filter((r) => signal(r.pgid === r.pid ? -r.pid : r.pid, sig));
|
|
224
|
+
/** A read that takes longer than `ms` is given up: a frozen tree must not wait on a slow `ps` for its SIGKILL. */
|
|
225
|
+
const within = (ms: number) => Promise.race([look(), Bun.sleep(ms).then(() => undefined)]);
|
|
226
|
+
// What leads a group of its own (an MCP server, a tool command) gets a SIGTERM of its own and the same grace period:
|
|
227
|
+
// a git process stopped by SIGKILL leaves its index lock behind.
|
|
228
|
+
const termed = new Set<string>();
|
|
229
|
+
const term = (rows: ProcRow[] | undefined) => {
|
|
230
|
+
for (const r of rows ? left(rows) : []) if (r.pgid === r.pid && !termed.has(key(r))) { termed.add(key(r)); signal(-r.pid, "SIGTERM"); }
|
|
231
|
+
};
|
|
232
|
+
try {
|
|
233
|
+
const first = await look();
|
|
234
|
+
if (!exited()) signal(-pid, "SIGTERM"); // a reaped leader's group id is not proven any more
|
|
235
|
+
term(first);
|
|
236
|
+
// The grace period, for the leader and for what it started: what they start meanwhile is recorded while its parent
|
|
237
|
+
// still runs.
|
|
238
|
+
for (const end = Date.now() + termMs; Date.now() < end; ) {
|
|
239
|
+
await Bun.sleep(100);
|
|
240
|
+
const rows = await look();
|
|
241
|
+
term(rows);
|
|
242
|
+
if (exited() && rows && !left(rows).length) break;
|
|
243
|
+
}
|
|
244
|
+
// Freeze, enumerate, kill: a stopped process starts nothing, so a read after the freeze sees all of it.
|
|
245
|
+
for (const end = Date.now() + killMs; ; ) {
|
|
246
|
+
const rows = await look();
|
|
247
|
+
const rest = rows ? left(rows) : undefined;
|
|
248
|
+
const strangers = rows ? unproven(rows) : [];
|
|
249
|
+
if (exited() && !strangers.length && (rest ? !rest.length : !found.size && groupGone(pid))) return;
|
|
250
|
+
if (Date.now() >= end) {
|
|
251
|
+
throw new Error(`owned child ${pid}: ${rest ? `${rest.length + strangers.length} process(es) of its group or below it still running${strangers.length ? `, ${strangers.length} not proven its own and left alone` : ""}` : "the process table cannot be read to confirm what it started is gone"}`);
|
|
252
|
+
}
|
|
253
|
+
// What is stopped is killed or continued, never left frozen: the group by whether its STOP was sent (its id stays
|
|
254
|
+
// reserved while a member lives), each process by whether the read after the freeze still shows it. What a read
|
|
255
|
+
// after the freeze shows for the first time (started between the read and the STOPs, in a group of its own) is
|
|
256
|
+
// frozen too, and read again, before anything is killed.
|
|
257
|
+
const groupStopped = !exited() && signal(-pid, "SIGSTOP");
|
|
258
|
+
const stopped = new Map<string, ProcRow>();
|
|
259
|
+
let frozen: ProcRow[] | undefined;
|
|
260
|
+
for (let todo = rest ?? [], round = 0; ; round++) {
|
|
261
|
+
for (const r of each(todo, "SIGSTOP")) stopped.set(key(r), r);
|
|
262
|
+
frozen = await within(1_000);
|
|
263
|
+
todo = frozen ? left(frozen).filter((r) => !stopped.has(key(r))) : [];
|
|
264
|
+
if (!todo.length || round >= 4) break;
|
|
265
|
+
}
|
|
266
|
+
// Without a read after the freeze, only what the STOP reached is killed: a stopped process keeps its pid.
|
|
267
|
+
const kill = frozen ? left(frozen) : [...stopped.values()];
|
|
268
|
+
each(kill, "SIGKILL");
|
|
269
|
+
if (groupStopped) signal(-pid, "SIGKILL");
|
|
270
|
+
for (const r of stopped.values()) if (!kill.some((k) => k.pid === r.pid && k.started === r.started)) signal(r.pgid === r.pid ? -r.pid : r.pid, "SIGCONT");
|
|
271
|
+
await Bun.sleep(50);
|
|
272
|
+
}
|
|
273
|
+
} finally {
|
|
274
|
+
dropPipes(proc); // a process that left the tree between two reads may still hold them: it must not keep the hub alive
|
|
275
|
+
}
|
|
276
|
+
}
|
|
@@ -0,0 +1,308 @@
|
|
|
1
|
+
import type { Task } from "./board.ts";
|
|
2
|
+
import type { PeerId } from "./envelope.ts";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Turn-free cohorts (issue #107): the owners of overlapping tasks, frozen into one group from the moment the overlap is
|
|
6
|
+
* found until each of them has stopped after its task closed. Whether a cohort is silent is decided when it is formed
|
|
7
|
+
* (turn-free on, no PII, every owner's context path verified) and never switched on later; it is lifted when an owner
|
|
8
|
+
* that cannot receive facts joins, a path is lost, or a PII task opens. Membership does not end when the board closes a
|
|
9
|
+
* task: a member stays in until its native turn has ended after that, so a late final answer is still the cohort's.
|
|
10
|
+
* That settlement is recorded when it happens and never undone: a settled member's later turns are new work.
|
|
11
|
+
*
|
|
12
|
+
* Completion is two-step. Each member's done is an intent. The member whose intent completes the set is selected, in
|
|
13
|
+
* one synchronous step, to integrate: it is asked to check its work against the others before its done is recorded,
|
|
14
|
+
* and its next done is accepted only for the same target (owner generation, cohort revision, files) once every other
|
|
15
|
+
* member has settled. Anything that changes the target asks again, a bounded number of times; past that the outcome is
|
|
16
|
+
* recorded as unresolved, never as integrated.
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
/** Integration requests for one cohort revision before the outcome is recorded as unresolved. */
|
|
20
|
+
export const MAX_REQUESTS = 3;
|
|
21
|
+
/** A done this soon after an integration request is a retry of a lost answer, not a check of the work. */
|
|
22
|
+
export const RETRY_MS = 2000;
|
|
23
|
+
|
|
24
|
+
export interface Member {
|
|
25
|
+
task: number;
|
|
26
|
+
owner: PeerId;
|
|
27
|
+
/** Changes whenever the task changes hands. */
|
|
28
|
+
gen: number;
|
|
29
|
+
/** When its owner was handed the task: its writes count for the integration target from here until it settles. */
|
|
30
|
+
since: number;
|
|
31
|
+
/** When the task left the open states for its owner. */
|
|
32
|
+
closedAt?: number;
|
|
33
|
+
/** When its owner's native turn first ended after that (or the task closed while the owner was idle). */
|
|
34
|
+
settledAt?: number;
|
|
35
|
+
/** When its owner's native turn first ended after its current intent: its writes for the task have stopped. */
|
|
36
|
+
stoppedAt?: number;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export interface Integration {
|
|
40
|
+
task: number;
|
|
41
|
+
owner: PeerId;
|
|
42
|
+
gen: number;
|
|
43
|
+
revision: number;
|
|
44
|
+
/** The files as they were at the last request: the next done is accepted only for this target. */
|
|
45
|
+
tree: string;
|
|
46
|
+
requests: number;
|
|
47
|
+
/** When the last request went out: a done right after it is a retry, not a confirmation. */
|
|
48
|
+
at: number;
|
|
49
|
+
/** The fact offer that went out with the last request: the next done acknowledges it. */
|
|
50
|
+
offer?: string;
|
|
51
|
+
confirmed?: boolean;
|
|
52
|
+
/** The outcome was recorded as unresolved: this revision asks nothing more, and the member's check counts as usual. */
|
|
53
|
+
closed?: boolean;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
export interface Cohort {
|
|
57
|
+
id: number;
|
|
58
|
+
/** Bumps on every change of membership or owner, when an intent is withdrawn, and when the cohort is lifted. */
|
|
59
|
+
revision: number;
|
|
60
|
+
members: Map<number, Member>;
|
|
61
|
+
silent: boolean;
|
|
62
|
+
intents: Map<number, { gen: number; at: number }>;
|
|
63
|
+
integration?: Integration;
|
|
64
|
+
/** Members whose completed-change notice the silence withheld: what replaces it if no integration runs. */
|
|
65
|
+
held: Set<number>;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
export interface CohortDeps {
|
|
69
|
+
/** Whether these owners may work without messages: turn-free on, no PII, every owner's context path verified. */
|
|
70
|
+
silence: (owners: PeerId[]) => boolean;
|
|
71
|
+
/**
|
|
72
|
+
* Whether `peer` is between native turns now (Codex not busy; Claude's last tool call started before its last Stop).
|
|
73
|
+
* Evidence is a native turn end, never a task approval or a delivery acknowledgement.
|
|
74
|
+
*/
|
|
75
|
+
idle: (peer: PeerId) => boolean;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
export type Completion =
|
|
79
|
+
| { action: "proceed"; integrated?: boolean }
|
|
80
|
+
| { action: "request"; why: string; requests: number; cohort: Cohort }
|
|
81
|
+
| { action: "unresolved"; why: string; cohort: Cohort };
|
|
82
|
+
|
|
83
|
+
export class Cohorts {
|
|
84
|
+
private readonly live: Cohort[] = [];
|
|
85
|
+
private next = 1;
|
|
86
|
+
|
|
87
|
+
constructor(private readonly d: CohortDeps) {}
|
|
88
|
+
|
|
89
|
+
/** The live cohort `task` is in. A cohort is over the moment its members have all finished, whoever asks. */
|
|
90
|
+
of(task: number): Cohort | undefined {
|
|
91
|
+
this.gc();
|
|
92
|
+
return this.live.find((c) => c.members.has(task));
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
list(): Cohort[] {
|
|
96
|
+
this.gc();
|
|
97
|
+
return [...this.live];
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* `task` overlaps `others` (open tasks of other owners): from now on they are one cohort. Merges cohorts the tasks were
|
|
102
|
+
* in, records owner changes (each bumps the revision and voids that member's intent), and lifts a silent cohort that
|
|
103
|
+
* an owner without verified facts joins. Returns what happened, for the notices.
|
|
104
|
+
*/
|
|
105
|
+
join(task: Task, others: Task[], gen: (t: Task) => number, handed: (t: Task) => number = () => Date.now()): { cohort: Cohort; formed: boolean; lifted: boolean } | undefined {
|
|
106
|
+
this.gc();
|
|
107
|
+
const all = [task, ...others].filter((t) => t.owner);
|
|
108
|
+
const found = [...new Set(all.map((t) => this.of(t.id)).filter((c): c is Cohort => !!c))];
|
|
109
|
+
if (!found.length && all.length < 2) return undefined;
|
|
110
|
+
const wasSilent = found.some((c) => c.silent);
|
|
111
|
+
let cohort = found[0];
|
|
112
|
+
const formed = !cohort;
|
|
113
|
+
if (!cohort) {
|
|
114
|
+
cohort = { id: this.next++, revision: 0, members: new Map(), silent: false, intents: new Map(), held: new Set() };
|
|
115
|
+
this.live.push(cohort);
|
|
116
|
+
}
|
|
117
|
+
let changed = formed;
|
|
118
|
+
for (const other of found.slice(1)) {
|
|
119
|
+
for (const [id, m] of other.members) cohort.members.set(id, m);
|
|
120
|
+
for (const [id, i] of other.intents) cohort.intents.set(id, i);
|
|
121
|
+
for (const id of other.held) cohort.held.add(id);
|
|
122
|
+
cohort.silent &&= other.silent;
|
|
123
|
+
this.live.splice(this.live.indexOf(other), 1);
|
|
124
|
+
changed = true;
|
|
125
|
+
}
|
|
126
|
+
for (const t of all) {
|
|
127
|
+
const m = cohort.members.get(t.id);
|
|
128
|
+
const g = gen(t);
|
|
129
|
+
if (m && m.owner === t.owner && m.gen === g) continue;
|
|
130
|
+
cohort.members.set(t.id, { task: t.id, owner: t.owner!, gen: g, since: handed(t) });
|
|
131
|
+
cohort.intents.delete(t.id);
|
|
132
|
+
changed = true;
|
|
133
|
+
}
|
|
134
|
+
if (changed) cohort.revision++;
|
|
135
|
+
const owners = [...new Set([...cohort.members.values()].map((m) => m.owner))];
|
|
136
|
+
if (formed) cohort.silent = this.d.silence(owners);
|
|
137
|
+
else if (cohort.silent && changed && !this.d.silence(owners)) cohort.silent = false;
|
|
138
|
+
// Lifted: a silent cohort (or a silent one merged into this) is no longer silent, and its members must hear it.
|
|
139
|
+
return { cohort, formed, lifted: wasSilent && !cohort.silent };
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
/** An owner lost its context path: every silent cohort it is in speaks again. */
|
|
143
|
+
lift(peer: PeerId): Cohort[] {
|
|
144
|
+
return this.liftWhere((c) => [...c.members.values()].some((m) => m.owner === peer));
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/** A PII task opened: every silent cohort speaks again, at once (issue #108). */
|
|
148
|
+
liftAll(): Cohort[] {
|
|
149
|
+
return this.liftWhere(() => true);
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
private liftWhere(pick: (c: Cohort) => boolean): Cohort[] {
|
|
153
|
+
this.gc(); // a finished cohort is not lifted: a peer leaving after the work (a benchmark's teardown) changes nothing
|
|
154
|
+
const out = this.live.filter((c) => c.silent && pick(c));
|
|
155
|
+
for (const c of out) {
|
|
156
|
+
c.silent = false;
|
|
157
|
+
c.revision++;
|
|
158
|
+
}
|
|
159
|
+
return out;
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* The silent cohort that makes a message from `from` to `to` cohort coordination: `from` is a member that has not
|
|
164
|
+
* settled since its task closed, and `to` is a member too. A settled member works on something else now.
|
|
165
|
+
*/
|
|
166
|
+
silenced(from: PeerId, to: PeerId): Cohort | undefined {
|
|
167
|
+
this.gc();
|
|
168
|
+
return this.live.find((c) => {
|
|
169
|
+
if (!c.silent) return false;
|
|
170
|
+
const members = [...c.members.values()];
|
|
171
|
+
return members.some((m) => m.owner === from && m.settledAt === undefined) && members.some((m) => m.owner === to);
|
|
172
|
+
});
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/** The task left the open states for its owner (done, approved, in review); an idle owner settles at once. */
|
|
176
|
+
closed(task: number, at = Date.now()): void {
|
|
177
|
+
const m = this.of(task)?.members.get(task);
|
|
178
|
+
if (!m) return;
|
|
179
|
+
m.closedAt ??= at;
|
|
180
|
+
if (m.settledAt === undefined && this.d.idle(m.owner)) m.settledAt = at;
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
/** `peer`'s native turn ended: its members whose tasks closed before settle, and those with an intent stop, for good. */
|
|
184
|
+
turnEnded(peer: PeerId, at = Date.now()): void {
|
|
185
|
+
for (const c of this.live) {
|
|
186
|
+
for (const m of c.members.values()) {
|
|
187
|
+
if (m.owner !== peer) continue;
|
|
188
|
+
if (m.closedAt !== undefined && m.closedAt <= at && m.settledAt === undefined) m.settledAt = at;
|
|
189
|
+
const intent = c.intents.get(m.task);
|
|
190
|
+
if (intent && intent.gen === m.gen && intent.at <= at && m.stoppedAt === undefined) m.stoppedAt = at;
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/**
|
|
196
|
+
* The task is open again for its owner (a failed check, changes requested, a reopen): its intent is void, and while
|
|
197
|
+
* its cohort is live it is that cohort's work again. A cohort that is over stays over.
|
|
198
|
+
*/
|
|
199
|
+
withdraw(task: number): void {
|
|
200
|
+
const c = this.of(task);
|
|
201
|
+
if (!c) return;
|
|
202
|
+
const m = c.members.get(task);
|
|
203
|
+
if (m) {
|
|
204
|
+
delete m.closedAt;
|
|
205
|
+
delete m.settledAt;
|
|
206
|
+
delete m.stoppedAt;
|
|
207
|
+
}
|
|
208
|
+
c.intents.delete(task);
|
|
209
|
+
c.revision++;
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
/**
|
|
213
|
+
* A done of `task` that selects nobody and asks nothing: the console finishing a member (issue #107). It still counts
|
|
214
|
+
* as that member's intent, so the member whose done completes the set integrates.
|
|
215
|
+
*/
|
|
216
|
+
intent(c: Cohort, task: Task, gen: number): void {
|
|
217
|
+
if (c.intents.has(task.id) && c.intents.get(task.id)!.gen === gen) return;
|
|
218
|
+
c.intents.set(task.id, { gen, at: Date.now() });
|
|
219
|
+
const m = c.members.get(task.id);
|
|
220
|
+
if (!m) return;
|
|
221
|
+
// An owner between turns (the console finished the task, say) has stopped already; one in a turn stops when it ends.
|
|
222
|
+
if (this.d.idle(m.owner)) m.stoppedAt = Date.now();
|
|
223
|
+
else delete m.stoppedAt;
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
/** Whether this done of the integrating member is a retry of a lost answer: then nothing counts or is acknowledged. */
|
|
227
|
+
isRetry(c: Cohort, task: Task, gen: number): boolean {
|
|
228
|
+
const ig = c.integration;
|
|
229
|
+
return !!ig && ig.task === task.id && ig.gen === gen && ig.owner === task.owner && ig.revision === c.revision && !ig.confirmed && Date.now() - ig.at < RETRY_MS;
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/**
|
|
233
|
+
* `task`'s owner called done in silent cohort `c`. Records its intent; then either it proceeds (others are still at
|
|
234
|
+
* work, another member integrates this revision, or its integration is confirmed), or it is asked to integrate (first
|
|
235
|
+
* request, or the target moved), or, past MAX_REQUESTS, the outcome is unresolved.
|
|
236
|
+
*/
|
|
237
|
+
completion(c: Cohort, task: Task, now: { gen: number; tree: string; factsCurrent: boolean; handed?: number }): Completion {
|
|
238
|
+
const m = c.members.get(task.id);
|
|
239
|
+
if (!m || m.owner !== task.owner || m.gen !== now.gen) {
|
|
240
|
+
c.members.set(task.id, { task: task.id, owner: task.owner!, gen: now.gen, since: now.handed ?? Date.now() });
|
|
241
|
+
c.revision++;
|
|
242
|
+
}
|
|
243
|
+
this.intent(c, task, now.gen);
|
|
244
|
+
const missing = [...c.members.values()].filter((x) => c.intents.get(x.task)?.gen !== x.gen);
|
|
245
|
+
if (missing.length) return { action: "proceed" };
|
|
246
|
+
const ig = c.integration;
|
|
247
|
+
if (ig && ig.revision === c.revision && ig.task !== task.id) return { action: "proceed" }; // another member integrates
|
|
248
|
+
if (!ig || ig.revision !== c.revision || ig.task !== task.id || ig.gen !== now.gen || ig.owner !== task.owner) {
|
|
249
|
+
c.integration = { task: task.id, owner: task.owner!, gen: now.gen, revision: c.revision, tree: now.tree, requests: 1, at: Date.now() };
|
|
250
|
+
return { action: "request", why: "", requests: 1, cohort: c };
|
|
251
|
+
}
|
|
252
|
+
if (ig.closed) return { action: "proceed" }; // recorded as unresolved: nothing more is asked of this revision
|
|
253
|
+
// Its next done: accepted only for the same target, once every other member has settled. A later turn of a settled
|
|
254
|
+
// member that touches these files moves the target, which the file hash catches.
|
|
255
|
+
// Other owners only: the integrating owner's own other tasks in the cohort stop with this very turn.
|
|
256
|
+
const running = [...c.members.values()].filter((x) => x.owner !== task.owner && x.stoppedAt === undefined).map((x) => x.owner);
|
|
257
|
+
let why = "";
|
|
258
|
+
if (running.length) why = `${[...new Set(running)].join(", ")} has not stopped since its done, so its changes may not be final`;
|
|
259
|
+
else if (now.tree !== ig.tree) why = "the files changed since the last request";
|
|
260
|
+
else if (!now.factsCurrent) why = "there are changes you have not been shown yet";
|
|
261
|
+
if (!why) {
|
|
262
|
+
ig.confirmed = true;
|
|
263
|
+
return { action: "proceed", integrated: true };
|
|
264
|
+
}
|
|
265
|
+
if (ig.requests >= MAX_REQUESTS) {
|
|
266
|
+
ig.closed = true;
|
|
267
|
+
return { action: "unresolved", why, cohort: c };
|
|
268
|
+
}
|
|
269
|
+
ig.requests++;
|
|
270
|
+
ig.tree = now.tree;
|
|
271
|
+
ig.at = Date.now();
|
|
272
|
+
delete ig.offer;
|
|
273
|
+
delete ig.confirmed; // asked again: a done right after this is a retry, and only a later one can confirm
|
|
274
|
+
return { action: "request", why, requests: ig.requests, cohort: c };
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
/**
|
|
278
|
+
* Whether a check that passed for `task` counts: always, unless `task` is the confirmed integrating member of its
|
|
279
|
+
* cohort's current revision, whose check counts only for the target it confirmed. A member that stopped being the
|
|
280
|
+
* last to finish (the revision moved) is an earlier finisher again, and its own check counts.
|
|
281
|
+
*/
|
|
282
|
+
holds(task: number, gen: number, tree: string): boolean {
|
|
283
|
+
const c = this.of(task);
|
|
284
|
+
const ig = c?.integration;
|
|
285
|
+
if (!c || !c.silent || !ig || ig.task !== task || ig.revision !== c.revision || ig.closed) return true;
|
|
286
|
+
return !!ig.confirmed && ig.gen === gen && ig.tree === tree;
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
/** The task lost its owner (released, unassigned): it leaves the cohort, and the revision moves on. */
|
|
290
|
+
leave(task: number): void {
|
|
291
|
+
const c = this.of(task);
|
|
292
|
+
if (!c) return;
|
|
293
|
+
c.members.delete(task);
|
|
294
|
+
c.intents.delete(task);
|
|
295
|
+
c.revision++;
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
/**
|
|
299
|
+
* A cohort is over once every member closed its task and settled (silent), or closed it (not silent: nothing is held
|
|
300
|
+
* back, so nothing waits for a turn end).
|
|
301
|
+
*/
|
|
302
|
+
private gc(): void {
|
|
303
|
+
for (const c of [...this.live]) {
|
|
304
|
+
const members = [...c.members.values()];
|
|
305
|
+
if (members.every((m) => (c.silent ? m.settledAt !== undefined : m.closedAt !== undefined))) this.live.splice(this.live.indexOf(c), 1);
|
|
306
|
+
}
|
|
307
|
+
}
|
|
308
|
+
}
|
|
@@ -6,10 +6,10 @@ export function stateDirFor(cwd: string): string {
|
|
|
6
6
|
return projectContext(cwd).stateDir;
|
|
7
7
|
}
|
|
8
8
|
|
|
9
|
-
/** Control WS wire version. 2 = `deliver` carries `envs` (digests); 3 = `tools` role and task messages; 4 = budget messages and `hub_checkpoint`; 5 = `ask`; 6 = console-only `ui` session bootstrap; 8 = controlled recovery; 9 = Pi bridge metadata; 10 = durable delivery receipts; 11 = queue hold diagnostics; 12 = generation-bound channel settlement and execution budgets. The plugin is installed apart from the daemon, so they can drift. */
|
|
10
|
-
export const PROTOCOL =
|
|
9
|
+
/** Control WS wire version. 2 = `deliver` carries `envs` (digests); 3 = `tools` role and task messages; 4 = budget messages and `hub_checkpoint`; 5 = `ask`; 6 = console-only `ui` session bootstrap; 8 = controlled recovery; 9 = Pi bridge metadata; 10 = durable delivery receipts; 11 = queue hold diagnostics; 12 = generation-bound channel settlement and execution budgets; 13 = turn-free facts (`facts` with tool, session and transcript binding, `silenced`) and per-recipient send results (issue #108). The plugin is installed apart from the daemon, so they can drift. */
|
|
10
|
+
export const PROTOCOL = 13;
|
|
11
11
|
/** Protocols a current coordinator may authenticate while upgrading a running source. */
|
|
12
|
-
export const RECOVERY_SOURCE_PROTOCOLS = [9, 10, 11, PROTOCOL] as const;
|
|
12
|
+
export const RECOVERY_SOURCE_PROTOCOLS = [9, 10, 11, 12, PROTOCOL] as const;
|
|
13
13
|
|
|
14
14
|
export interface Hello {
|
|
15
15
|
/** `tools`: acts for `peer` (task tools, hub_send) without being a delivery target: the MCP server Kimi and Codex run. */
|