viber-channel 0.8.19 → 0.8.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,521 @@
1
+ /**
2
+ * runner_roster.ts — the machine's presence, as the runner reports it (#627 step-03).
3
+ *
4
+ * Each runner heartbeat carries a ROSTER built from the agents' presence records
5
+ * (`presence_record.ts`) and ONE enumeration of the machine's processes:
6
+ *
7
+ * present the record is active and its pid still has the SAME start time
8
+ * closed the record says the agent ended on a DECIDED cause (see `endVerdict`)
9
+ * down the agent is dead without a decided end: crash, reboot, window closed…
10
+ *
11
+ * Nothing is reported when the runner cannot tell — never a guess:
12
+ * - the enumeration failed its positive control (it must contain the runner itself);
13
+ * - the record is unreadable, or its start time is not known yet;
14
+ * - the directory could not be read (then NO closed/down at all this beat).
15
+ *
16
+ * ⚠ The END POLICY lives here and nowhere else (plan 627, step-01 review): an agent
17
+ * records a fact, the runner decides. JP's rule: grey rather than lose.
18
+ *
19
+ * A closed/down record is resent on every beat until the server acknowledges it
20
+ * (`presence_applied: true` in a 2xx): only then is the file removed. So a failed
21
+ * beat, or a server that has not applied it, loses nothing.
22
+ */
23
+ import { mkdirSync, readdirSync, renameSync, unlinkSync, writeFileSync } from "node:fs";
24
+ import { dirname, join } from "node:path";
25
+ import {
26
+ type PresenceEndReason,
27
+ type PresenceRecord,
28
+ presenceDir,
29
+ RECORD_FILE_RE,
30
+ readPresenceRecord,
31
+ } from "./presence_record.js";
32
+ import { enumerateStartTimes } from "./process_start.js";
33
+ import { runnerIdentity } from "./runner_exec.js";
34
+ import * as reg from "./runner_registry.js";
35
+ import type { SeenAgent } from "./runner_registry.js";
36
+
37
+ /** Most instances one beat may carry; the server refuses more (step-04). */
38
+ export const ROSTER_MAX = 200;
39
+
40
+ /**
41
+ * The largest heartbeat body the server accepts (step-04 sets it on the route).
42
+ * Derived, not guessed: 200 entries at their maximum size must fit — pinned by a
43
+ * test that builds exactly that body. The historical limit was 8 KiB, which about
44
+ * 90 agents would overflow, taking the RUNNER's own beat down with them (Opus).
45
+ */
46
+ export const HEARTBEAT_MAX_BODY_BYTES = 32 * 1024;
47
+
48
+ export type EndVerdict = "closed" | "down";
49
+
50
+ /**
51
+ * THE end policy. Closed (removed from the web) only for a cause somebody decided;
52
+ * every ambiguous end — a pipe closed, a signal, a failure — is down (greyed,
53
+ * relaunchable). A Windows reboot closes claude's stdin exactly like `/exit` does:
54
+ * that is why `stdin-end` is down.
55
+ */
56
+ export function endVerdict(reason: PresenceEndReason): EndVerdict {
57
+ switch (reason) {
58
+ case "revoked":
59
+ case "lease-lost":
60
+ case "once-completed":
61
+ return "closed";
62
+ case "stdin-end":
63
+ case "signal":
64
+ case "shutdown":
65
+ return "down";
66
+ }
67
+ }
68
+
69
+ export interface RosterPresent {
70
+ instance_id: string;
71
+ conversation_id: string | null;
72
+ /** The process start time the runner verified. Two present instances sharing a
73
+ * label (a relaunch overlapping its old process): Python lets the LATEST win. */
74
+ started_at: number;
75
+ }
76
+
77
+ export interface Roster {
78
+ present: RosterPresent[];
79
+ closed: string[];
80
+ down: string[];
81
+ }
82
+
83
+ export interface CollectedRoster {
84
+ roster: Roster;
85
+ /** The live records, for the registry join (`syncFromRecords`). */
86
+ seen: SeenAgent[];
87
+ /** Record file of each closed/down instance, with the EXACT content that was read:
88
+ * removed once acknowledged, and only if the file still holds that content. */
89
+ endedFiles: Map<string, { path: string; content: string }>;
90
+ /** Why this beat reported nothing it could not prove. */
91
+ notes: string[];
92
+ }
93
+
94
+ /** The process table: pid → start time, or `undefined` when the instrument did not see. */
95
+ export type ProcessTable = Map<number, number | undefined>;
96
+
97
+ export interface CollectDeps {
98
+ identityBaseUrl: string;
99
+ lockDir?: string;
100
+ /** ONE enumeration of the machine (`enumerateStartTimes`); undefined = not seen. */
101
+ enumerate: () => Promise<ProcessTable | undefined>;
102
+ }
103
+
104
+ function emptyCollected(note?: string): CollectedRoster {
105
+ return {
106
+ roster: { present: [], closed: [], down: [] },
107
+ seen: [],
108
+ endedFiles: new Map(),
109
+ notes: note === undefined ? [] : [note],
110
+ };
111
+ }
112
+
113
+ type Classified =
114
+ | { kind: "present"; record: PresenceRecord }
115
+ | { kind: "closed" | "down"; record: PresenceRecord }
116
+ | { kind: "unknown"; why: string };
117
+
118
+ /** Classify ONE record against the process table. PURE — the witness's subject. */
119
+ export function classifyRecord(record: PresenceRecord, table: ProcessTable): Classified {
120
+ if (record.state === "ended") {
121
+ // A decided or ambiguous END is a fact written by the agent itself: it holds
122
+ // even if the process is still unwinding.
123
+ return { kind: endVerdict(record.ended_reason as PresenceEndReason), record };
124
+ }
125
+ if (record.start_time_ms === null) return { kind: "unknown", why: `${record.instance_id}: start time not known yet` };
126
+ // Absent from the table → dead. Sound ONLY because the table passed its positive
127
+ // control (`enumerateStartTimes` saw the runner itself): a partial table never gets here.
128
+ if (!table.has(record.pid)) return { kind: "down", record };
129
+ const observed = table.get(record.pid);
130
+ if (observed === undefined) return { kind: "unknown", why: `${record.instance_id}: pid ${record.pid} has no readable start time` };
131
+ // ZERO tolerance: both numbers come from the same CIM expression (process_start.ts).
132
+ return observed === record.start_time_ms ? { kind: "present", record } : { kind: "down", record };
133
+ }
134
+
135
+ /** Read every record of this backend and classify it. Never throws. */
136
+ export async function collectRoster(deps: CollectDeps): Promise<CollectedRoster> {
137
+ const dir = presenceDir(deps.identityBaseUrl, deps.lockDir);
138
+ let names: string[];
139
+ try {
140
+ names = readdirSync(dir).filter((n) => RECORD_FILE_RE.test(n));
141
+ } catch (err) {
142
+ if ((err as NodeJS.ErrnoException).code === "ENOENT") return emptyCollected();
143
+ return emptyCollected(`presence directory unreadable (${String(err)}): nothing reported`);
144
+ }
145
+ if (names.length === 0) return emptyCollected();
146
+
147
+ const table = await deps.enumerate();
148
+ if (table === undefined) return emptyCollected("process enumeration failed its control: nothing reported");
149
+
150
+ const out = emptyCollected();
151
+ for (const name of names) {
152
+ const path = join(dir, name);
153
+ const read = readPresenceRecord(path);
154
+ if (read.kind !== "ok") {
155
+ if (read.kind === "unreadable") out.notes.push(`${name}: unreadable (${read.error})`);
156
+ continue;
157
+ }
158
+ const c = classifyRecord(read.record, table);
159
+ if (c.kind === "unknown") {
160
+ out.notes.push(c.why);
161
+ continue;
162
+ }
163
+ const { record } = c;
164
+ if (c.kind === "present") {
165
+ out.roster.present.push({
166
+ instance_id: record.instance_id,
167
+ conversation_id: record.conversation_id,
168
+ started_at: record.start_time_ms as number, // non-null: classifyRecord checked it
169
+ });
170
+ out.seen.push({
171
+ label: record.label,
172
+ project_id: record.project_id,
173
+ instance_id: record.instance_id,
174
+ conversation_id: record.conversation_id,
175
+ created_at: record.created_at,
176
+ job_name: record.job_name,
177
+ });
178
+ } else {
179
+ out.roster[c.kind].push(record.instance_id);
180
+ out.endedFiles.set(record.instance_id, { path, content: JSON.stringify(record) });
181
+ }
182
+ }
183
+ return capRoster(out);
184
+ }
185
+
186
+ /** Keep the beat under `ROSTER_MAX`: ended entries first (they must not be lost), then present. */
187
+ function capRoster(c: CollectedRoster): CollectedRoster {
188
+ const { present, closed, down } = c.roster;
189
+ const total = present.length + closed.length + down.length;
190
+ if (total <= ROSTER_MAX) return c;
191
+ let room = ROSTER_MAX;
192
+ const keptClosed = closed.slice(0, room);
193
+ room -= keptClosed.length;
194
+ const keptDown = down.slice(0, room);
195
+ room -= keptDown.length;
196
+ const keptPresent = present.slice(0, room);
197
+ const kept = new Set([...keptClosed, ...keptDown]);
198
+ for (const id of [...c.endedFiles.keys()]) if (!kept.has(id)) c.endedFiles.delete(id);
199
+ // The registry must only learn what this beat actually told the server (Codex).
200
+ const sentPresent = new Set(keptPresent.map((p) => p.instance_id));
201
+ c.seen = c.seen.filter((s) => sentPresent.has(s.instance_id));
202
+ c.roster = { present: keptPresent, closed: keptClosed, down: keptDown };
203
+ c.notes.push(`roster capped at ${ROSTER_MAX} (had ${total}); the rest goes next beat`);
204
+ return c;
205
+ }
206
+
207
+ /** The server's answer to a roster beat (step-04 contract). */
208
+ export interface RosterAck {
209
+ /** true only when Python applied the presence; false/absent → resend next beat. */
210
+ presence_applied: boolean;
211
+ /** The server's id of this beat: the identifier the agents count (step-06). */
212
+ beat_id: string | null;
213
+ /** Lease state per present instance: true / false / null (indeterminate). */
214
+ instances: Map<string, boolean | null>;
215
+ /** #627 (decision A): replaced instances whose conversations are not handed over yet. */
216
+ handover_pending: Set<string>;
217
+ }
218
+
219
+ /** Parse a 2xx body. Anything unexpected reads as "not applied, no lease info". */
220
+ export function parseRosterAck(body: unknown): RosterAck {
221
+ const ack: RosterAck = { presence_applied: false, beat_id: null, instances: new Map(), handover_pending: new Set() };
222
+ if (typeof body !== "object" || body === null) return ack;
223
+ const b = body as Record<string, unknown>;
224
+ ack.presence_applied = b.presence_applied === true;
225
+ ack.beat_id = typeof b.beat_id === "string" && b.beat_id !== "" ? b.beat_id : null;
226
+ if (Array.isArray(b.handover_pending)) {
227
+ for (const id of b.handover_pending) if (typeof id === "string") ack.handover_pending.add(id);
228
+ }
229
+ if (typeof b.instances === "object" && b.instances !== null) {
230
+ for (const [id, v] of Object.entries(b.instances as Record<string, unknown>)) {
231
+ const lease = typeof v === "object" && v !== null ? (v as Record<string, unknown>).lease_active : undefined;
232
+ if (lease === true || lease === false || lease === null) ack.instances.set(id, lease);
233
+ }
234
+ }
235
+ return ack;
236
+ }
237
+
238
+ /** One instance's coverage, read by its agent (step-06). */
239
+ export interface Coverage {
240
+ schema_version: 1;
241
+ instance_id: string;
242
+ lease_active: boolean | null;
243
+ /** EXACTLY the server's value: the agent counts each id once. */
244
+ beat_id: string;
245
+ /** The conversation the lease was renewed for. */
246
+ conversation_id: string | null;
247
+ /** LOCAL clock of this write: freshness is judged on it, never on beat_id. */
248
+ written_at: number;
249
+ }
250
+
251
+ /** `<instance_id>.coverage.json` — does not match `RECORD_FILE_RE`, by construction. */
252
+ export function coveragePath(identityBaseUrl: string, instanceId: string, lockDir?: string): string {
253
+ return join(presenceDir(identityBaseUrl, lockDir), `${instanceId}.coverage.json`);
254
+ }
255
+
256
+ function atomicWrite(path: string, value: unknown): void {
257
+ mkdirSync(dirname(path), { recursive: true });
258
+ const tmp = `${path}.${process.pid}.tmp`;
259
+ writeFileSync(tmp, JSON.stringify(value), "utf8");
260
+ try {
261
+ renameSync(tmp, path);
262
+ } catch (err) {
263
+ try {
264
+ unlinkSync(tmp);
265
+ } catch {
266
+ // already gone
267
+ }
268
+ throw err;
269
+ }
270
+ }
271
+
272
+ export interface ApplyAckDeps {
273
+ identityBaseUrl: string;
274
+ lockDir?: string;
275
+ now: () => number;
276
+ log: (line: string) => void;
277
+ }
278
+
279
+ /**
280
+ * After a 2xx: write each present instance's coverage (when the server gave its
281
+ * lease and a beat id), and — only when the presence was APPLIED — remove the
282
+ * acknowledged closed/down records. Returns the ids whose record was removed.
283
+ */
284
+ export function applyAck(collected: CollectedRoster, ack: RosterAck, deps: ApplyAckDeps): string[] {
285
+ if (ack.beat_id !== null) {
286
+ for (const p of collected.roster.present) {
287
+ if (!ack.instances.has(p.instance_id)) continue;
288
+ const coverage: Coverage = {
289
+ schema_version: 1,
290
+ instance_id: p.instance_id,
291
+ lease_active: ack.instances.get(p.instance_id) ?? null,
292
+ beat_id: ack.beat_id,
293
+ conversation_id: p.conversation_id,
294
+ written_at: deps.now(),
295
+ };
296
+ try {
297
+ atomicWrite(coveragePath(deps.identityBaseUrl, p.instance_id, deps.lockDir), coverage);
298
+ } catch (err) {
299
+ deps.log(`[runner] coverage write failed for ${p.instance_id}: ${String(err)}`);
300
+ }
301
+ }
302
+ }
303
+ if (!ack.presence_applied) return [];
304
+ const removed: string[] = [];
305
+ for (const [id, { path, content }] of collected.endedFiles) {
306
+ // The file may have been rewritten while the beat was in flight: remove only
307
+ // the version the server acknowledged. Compared on the WHOLE record, not on
308
+ // `written_at`, which two writes can share within one millisecond (Codex).
309
+ const now = readPresenceRecord(path);
310
+ if (now.kind !== "ok" || JSON.stringify(now.record) !== content) continue;
311
+ try {
312
+ unlinkSync(path);
313
+ removed.push(id);
314
+ } catch (err) {
315
+ deps.log(`[runner] could not remove acknowledged record ${id}: ${String(err)}`);
316
+ }
317
+ try {
318
+ unlinkSync(coveragePath(deps.identityBaseUrl, id, deps.lockDir));
319
+ } catch {
320
+ // no coverage for it: nothing to clean
321
+ }
322
+ }
323
+ return removed;
324
+ }
325
+
326
+ export interface RosterBeatAuth {
327
+ base_url: string;
328
+ runner_id: string;
329
+ runner_token: string;
330
+ identity_base_url?: string;
331
+ }
332
+
333
+ export interface RosterBeatDeps {
334
+ fetchImpl?: typeof fetch;
335
+ enumerate?: () => Promise<ProcessTable | undefined>;
336
+ lockDir?: string;
337
+ now?: () => number;
338
+ log?: (line: string) => void;
339
+ headers?: () => Record<string, string>;
340
+ }
341
+
342
+ /**
343
+ * ONE runner heartbeat carrying the roster, and everything a 2xx makes true locally:
344
+ * coverage files, acknowledged records removed, the registry kept in step, and
345
+ * `runner.json`. Returns "revoked" on a 401 so the daemon tears down.
346
+ *
347
+ * A failed or non-2xx beat changes NOTHING on disk: the coverage then expires on its
348
+ * own and the ended records go again next beat.
349
+ */
350
+ export async function runRosterBeat(auth: RosterBeatAuth, deps: RosterBeatDeps = {}): Promise<"ok" | "revoked"> {
351
+ // SINGLE-FLIGHT. The enumeration alone can take up to 30 s — one heartbeat period —
352
+ // so an overlapping beat is ordinary, and two answers arriving in reverse order
353
+ // would rewrite an older coverage over a newer one (Codex, step-03 review). A tick
354
+ // that finds a beat in flight is skipped: the running one covers it.
355
+ if (beatInFlight) {
356
+ noteOnce("a heartbeat was skipped: the previous one is still in flight", deps.log ?? ((l: string) => process.stderr.write(`${l}\n`)));
357
+ return "ok";
358
+ }
359
+ beatInFlight = true;
360
+ try {
361
+ return await rosterBeatOnce(auth, deps);
362
+ } finally {
363
+ beatInFlight = false;
364
+ }
365
+ }
366
+ let beatInFlight = false;
367
+ let unappliedBeats = 0;
368
+ /** #627 (decision A): pending beats before the runner says the handover is slow. */
369
+ export const HANDOVER_WARN_AFTER = 10;
370
+ const handoverAttempts = new Map<string, number>();
371
+ /** Consecutive unacknowledged beats (with ended records) before the runner says so. */
372
+ export const UNAPPLIED_WARN_AFTER = 10;
373
+
374
+ async function rosterBeatOnce(auth: RosterBeatAuth, deps: RosterBeatDeps): Promise<"ok" | "revoked"> {
375
+ const log = deps.log ?? ((l: string) => process.stderr.write(`${l}\n`));
376
+ const now = deps.now ?? Date.now;
377
+ const identity = runnerIdentity(auth);
378
+ const collected = await collectRoster({
379
+ identityBaseUrl: identity,
380
+ lockDir: deps.lockDir,
381
+ enumerate: deps.enumerate ?? (() => enumerateStartTimes()),
382
+ });
383
+ for (const note of collected.notes) noteOnce(note, log);
384
+ // #627 step-07: an instance REPLACED by a relaunch is reported closed until the
385
+ // server acknowledges it — so no greyed ghost stays behind the new agent.
386
+ const { closes: pendingCloses, replaced } = pendingClosesOf(identity, deps.lockDir, collected.roster);
387
+ collected.roster.closed.push(...pendingCloses);
388
+
389
+ let resp: Response;
390
+ try {
391
+ resp = await (deps.fetchImpl ?? fetch)(`${auth.base_url}/api/runners/${auth.runner_id}/heartbeat`, {
392
+ method: "POST",
393
+ // never follow a redirect while carrying the bearer (Codex review, #460)
394
+ redirect: "error",
395
+ headers: {
396
+ Authorization: `Bearer ${auth.runner_token}`,
397
+ "Content-Type": "application/json",
398
+ ...(deps.headers?.() ?? {}),
399
+ },
400
+ body: JSON.stringify({
401
+ capabilities: { runtimes: ["claude-code", "codex", "gemma"], platform: process.platform, version: 1 },
402
+ // #627 (JP, decision A): the new instance of each pair takes the old one's
403
+ // conversations — the server re-invites it once it is online.
404
+ roster: replaced.length > 0 ? { ...collected.roster, replaced } : collected.roster,
405
+ }),
406
+ });
407
+ } catch {
408
+ return "ok"; // best-effort: nothing acknowledged, nothing changed
409
+ }
410
+ if (resp.status === 401) return "revoked";
411
+ if (resp.status < 200 || resp.status >= 300) {
412
+ // Not swallowed: a 4xx here means the server refused the beat — the RUNNER itself
413
+ // then goes offline after 90 s (Opus, step-03 review).
414
+ noteOnce(`heartbeat refused: HTTP ${resp.status}`, log);
415
+ return "ok";
416
+ }
417
+
418
+ const ack = parseRosterAck(await resp.json().catch(() => null));
419
+ // A beat whose ended records are never acknowledged is not dangerous (coverage and
420
+ // leases are still served) but it is SILENT: say it once it lasts (Opus, step-05).
421
+ if (collected.endedFiles.size > 0 && !ack.presence_applied) {
422
+ unappliedBeats++;
423
+ if (unappliedBeats === UNAPPLIED_WARN_AFTER) {
424
+ log(`[runner] roster not applied by the server for ${unappliedBeats} beats in a row — ended records are kept and resent`);
425
+ }
426
+ } else {
427
+ unappliedBeats = 0;
428
+ }
429
+ const removed = applyAck(collected, ack, { identityBaseUrl: identity, lockDir: deps.lockDir, now, log });
430
+ const closedNow = removed.filter((id) => collected.roster.closed.includes(id));
431
+ const result = reg.updateRegistry(reg.registryPath(identity, deps.lockDir), (r) => {
432
+ const synced = reg.syncFromRecords(r, collected.seen);
433
+ const closed = reg.markClosed(r, closedNow, now());
434
+ let cleared = false;
435
+ if (ack.presence_applied) {
436
+ for (const e of Object.values(r.entries)) {
437
+ // An older predecessor (handover_from) is done once the server no longer
438
+ // answers it pending — same rule as `pending_close` below.
439
+ if (e.handover_from?.length) {
440
+ const kept = e.handover_from.filter((id) => !pendingCloses.includes(id) || ack.handover_pending.has(id));
441
+ if (kept.length !== e.handover_from.length) {
442
+ e.handover_from = kept;
443
+ cleared = true;
444
+ }
445
+ }
446
+ if (e.pending_close && pendingCloses.includes(e.pending_close)) {
447
+ const id = e.pending_close;
448
+ // #627 (decision A): the server could not yet hand this agent's conversations
449
+ // to its replacement — keep the close and resend the pair (Opus I1). NEVER given
450
+ // up (Codex): a dropped pair would leave the conversations with nobody and no way
451
+ // to retry. It is bounded by the replacement's life: the pair travels only while
452
+ // the new agent is present. A long wait is said once, not silently absorbed.
453
+ if (ack.handover_pending.has(id)) {
454
+ const n = (handoverAttempts.get(id) ?? 0) + 1;
455
+ handoverAttempts.set(id, n);
456
+ if (n === HANDOVER_WARN_AFTER) {
457
+ log(`[runner] relaunch: conversations of ${id.slice(0, 8)} still not handed over after ${n} beats — retrying`);
458
+ }
459
+ continue;
460
+ }
461
+ handoverAttempts.delete(id);
462
+ e.pending_close = null;
463
+ cleared = true;
464
+ }
465
+ }
466
+ }
467
+ return synced || closed || cleared;
468
+ });
469
+ if (!result.ok) noteOnce(`registry not updated: ${result.error}`, log);
470
+ try {
471
+ reg.writeRunnerState(reg.runnerStatePath(identity, deps.lockDir), { runner_id: auth.runner_id, last_beat_ok_at: now() });
472
+ } catch (err) {
473
+ log(`[runner] runner.json write failed: ${String(err)}`);
474
+ }
475
+ return "ok";
476
+ }
477
+
478
+ /** A note logged at every 30 s beat would drown the log: once per distinct text. */
479
+ const loggedNotes = new Set<string>();
480
+ function noteOnce(note: string, log: (line: string) => void): void {
481
+ if (loggedNotes.has(note)) return;
482
+ if (loggedNotes.size > 500) loggedNotes.clear();
483
+ loggedNotes.add(note);
484
+ log(`[runner] roster: ${note}`);
485
+ }
486
+
487
+ /** One relaunch: the instance it replaced, and the one that now runs. */
488
+ export interface ReplacedPair {
489
+ old: string;
490
+ new: string;
491
+ }
492
+
493
+ /**
494
+ * Replaced instances still to report closed — never one the roster already carries.
495
+ * #627 (JP, decision A): a close waits until its REPLACEMENT is present in this same
496
+ * roster, and travels with the pair, so the server can hand the old instance's
497
+ * conversations to the new one at the one moment the old is retired.
498
+ */
499
+ export function pendingClosesOf(
500
+ identity: string,
501
+ lockDir: string | undefined,
502
+ roster: Roster,
503
+ ): { closes: string[]; replaced: ReplacedPair[] } {
504
+ const read = reg.readRegistry(reg.registryPath(identity, lockDir));
505
+ if (read.kind !== "ok") return { closes: [], replaced: [] };
506
+ const present = new Set(roster.present.map((p) => p.instance_id));
507
+ const carried = new Set([...present, ...roster.closed, ...roster.down]);
508
+ const room = Math.max(0, ROSTER_MAX - carried.size);
509
+ const closes: string[] = [];
510
+ const replaced: ReplacedPair[] = [];
511
+ for (const e of Object.values(read.registry.entries)) {
512
+ const next = e.instance_id;
513
+ if (!next || !present.has(next)) continue; // its replacement is not running yet
514
+ for (const id of [e.pending_close, ...(e.handover_from ?? [])]) {
515
+ if (!id || id === next || carried.has(id) || closes.includes(id) || closes.length >= room) continue;
516
+ closes.push(id);
517
+ replaced.push({ old: id, new: next });
518
+ }
519
+ }
520
+ return { closes, replaced };
521
+ }