@byok-sdk/server 0.12.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/hub.d.ts DELETED
@@ -1,947 +0,0 @@
1
- import type { WebSocket } from 'ws';
2
- import { type Envelope, type AgentMessagePublishPayload, type AgentMessageServerContext, type AgentHomeProjectionCompletionRequest, type AgentHomeProjectionReadback, type RuntimeId, type RuntimeInfo, type TaskState, type ToolsetId } from '@byok-sdk/protocol';
3
- import type { DeviceRegistry } from './auth';
4
- import { RateLimiter } from './rate-limiter';
5
- import type { TaskStore } from './task-store';
6
- import type { ByokServerEvent, AgentContentReadRequest, AgentHomeProjectionRequest, AgentEgressReceipt, DispatchInput, FreshAgentEgressDispatchInput, HubStats, MachineInfo, TaskHandle, TaskSnapshot } from './types';
7
- /**
8
- * The connection hub: tracks each device's live transport (WS or long-poll —
9
- * never both at once, see {@link takeOverAsLongPoll}), routes `dispatch()`'d
10
- * tasks to a device, and processes inbound task.* envelopes from daemons.
11
- *
12
- * Routing (M1): every `task.*` envelope carries a *required* envelope
13
- * `task_id` — the sole routing key, both directions. No payload duplicates
14
- * it, and none of the handlers below need to guard against a missing one
15
- * (the wire schema already rejects such an envelope before it reaches here).
16
- *
17
- * Inbound gate ({@link handleInbound}): the single choke point both
18
- * transports (`ws-server.ts`'s WS message handler, `http.ts`'s
19
- * `POST /byok/messages`) call instead of reaching into per-type handlers
20
- * directly. Runs, in order: (1) type-allow — only `DAEMON_TO_SERVER_TYPES`
21
- * plus the authenticated long-poll `conn.hello` snapshot may pass, a server
22
- * -> daemon type arriving inbound is rejected (P2); (2)
23
- * ownership (N2) — an envelope for a task already owned by a *different*
24
- * device is dropped and logged, never force-failed (force-failing on an
25
- * authz mismatch would let an attacker who merely guesses a `taskId` kill
26
- * the real owner's task); (3) dedup (N3) — an envelope `id` already seen
27
- * from this device is a no-op, making the at-least-once wire (§9)
28
- * effectively at-most-once server-side; (4) dispatch to the per-type
29
- * handler. Because ownership is enforced once, centrally, here, the
30
- * handlers below no longer carry their own device-mismatch checks.
31
- *
32
- * Outbound delivery (M1, §1.2/§9): every server -> daemon envelope
33
- * (`conn.ack`, either task offer, `task.approve/reject/cancel/steer`) gets a fresh
34
- * per-device monotonic `seq` and is retained in a capped ring buffer
35
- * ({@link OUTBOX_RING_CAPACITY} entries) so it can be redelivered — in `seq`
36
- * order, skipping anything whose task has since reached a terminal state —
37
- * on reconnect (`redeliverAfterReconnect`) or long-poll
38
- * (`pollEvents`/`collectRelevant`). `conn.ack` has no task association, so
39
- * it's retained (for seq-counting purposes only) but never redelivered.
40
- * Exception (N1/F4): `task.cancel`/`task.reject` are exempt from the
41
- * terminal-task skip (`OutboxEntry.redeliverThroughTerminal`) because both
42
- * move their own task to a terminal state before being queued — without the
43
- * exemption they could never qualify for redelivery even when the original
44
- * send never reached the daemon.
45
- *
46
- * State-machine (M1): `task.claim` only claims (`Offered -> Claimed`); it no
47
- * longer implies `Running`. The daemon reports `Claimed -> Running`
48
- * explicitly via `task.started` once its runtime session actually starts
49
- * (§3.1). `task.decline` reports a pre-claim fail-closed rejection
50
- * (`Offered -> Failed`, §3.2). `task.cancelled` is dual-purpose: an
51
- * idempotent ack when the server already cancelled the task itself, or the
52
- * authoritative trigger when the daemon observed the cancellation first
53
- * (§3.3). Per §9, `task.complete`/`task.fail`/`task.cancelled` arriving for
54
- * an already-terminal task are silently dropped as stale/duplicate — not a
55
- * warning; this is what naturally resolves the M0 gatekeeper's cancel-race
56
- * `console.warn` finding (a late `task.fail`/`task.cancelled` racing a
57
- * server-initiated cancel is exactly this case).
58
- *
59
- * Task lease (M2): a periodic sweep (started in the constructor) reaps a
60
- * `Claimed`/`Running`/`AwaitApproval` task to
61
- * `Failed(retryable: true, reason: 'lease-expired')` once its owning device
62
- * has been dark (disconnected, or long-poll-silent) AND the task itself has
63
- * had no inbound activity for `taskLeaseMs` — see the "task-lease reaper"
64
- * section further down for the full design, including why this does not
65
- * reintroduce the disconnect-alone-fails-the-task bug M1 removed above.
66
- */
67
- /**
68
- * M4 Phase 3: thrown by {@link ConnectionHub.approveTask}/{@link
69
- * ConnectionHub.rejectTask} for a `taskId` this hub has no record of at all
70
- * — mirrors `task-store.ts`'s `IllegalTaskTransitionError` (typed error
71
- * class + `instanceof` dispatch is this codebase's own established idiom for
72
- * mapping a domain error to the right status code). Distinguished from
73
- * {@link TaskNotAwaitingApprovalError} so a caller CAN tell the two failure
74
- * modes apart (e.g. 404 vs. 409) instead of only ever seeing a single
75
- * generic `Error`. There is no bearer-authed HTTP route for this on
76
- * `http.ts`'s own app (see that file's own closing comment for why) — the
77
- * supported entry point is calling `approveTask`/`rejectTask` directly, or
78
- * via `TaskHandle.approve()`/`reject()` (thin wrappers over the same two
79
- * methods); an embedder builds its own operator-facing surface on top of
80
- * that, exactly like `examples/basic/server.ts`'s own
81
- * `/api/tasks/:taskId/approve`/`reject` routes do.
82
- */
83
- export declare class UnknownTaskError extends Error {
84
- readonly taskId: string;
85
- constructor(taskId: string);
86
- }
87
- /**
88
- * Thrown by {@link ConnectionHub.approveTask}/{@link ConnectionHub.rejectTask}
89
- * when the task exists but isn't currently `AwaitApproval` — see {@link
90
- * UnknownTaskError}'s own doc comment for why this is a distinct class.
91
- * `verb` keeps the exact pre-existing message wording per call site
92
- * ("cannot approve ..." vs. "cannot reject ...") byte-for-byte unchanged —
93
- * this message is user-visible today (e.g. `examples/basic`'s own
94
- * `/api/tasks/:taskId/approve` surfaces `err.message` straight to the
95
- * caller), so only the error's TYPE changes here, not its text.
96
- */
97
- export declare class TaskNotAwaitingApprovalError extends Error {
98
- readonly taskId: string;
99
- readonly state: TaskState;
100
- constructor(taskId: string, state: TaskState, verb: 'approve' | 'reject');
101
- }
102
- /**
103
- * M5 (approval targeting): thrown by {@link ConnectionHub.approveTask}/
104
- * {@link ConnectionHub.rejectTask} when an operator-supplied
105
- * `opts.approvalId` is provided but does NOT match this hub's own
106
- * last-recorded {@link TaskSnapshot.pendingApprovalId} for `taskId` — i.e.
107
- * the caller is targeting a SPECIFIC approval that this server's record
108
- * shows has already been superseded by a newer one (see `onAwaitApproval`'s
109
- * own doc comment for how that happens). Distinct from
110
- * {@link TaskNotAwaitingApprovalError} (the task isn't `AwaitApproval` at
111
- * all right now — checked FIRST, so it still wins when both would apply):
112
- * this error means the task genuinely IS awaiting approval, just not the one
113
- * the caller thinks it is. Thrown before any state change and before any
114
- * wire message is sent — a stale-id call has zero side effects, same as any
115
- * other validation failure in this file.
116
- */
117
- export declare class StaleApprovalError extends Error {
118
- readonly taskId: string;
119
- readonly requestedApprovalId: string;
120
- readonly currentApprovalId: string | undefined;
121
- constructor(taskId: string, requestedApprovalId: string, currentApprovalId: string | undefined);
122
- }
123
- /**
124
- * S0 (GAP-002): why {@link ConnectionHub.steerTask} refused. Stable strings —
125
- * a caller (an embedder's operator UI, an HTTP surface mapping this to a
126
- * status code) switches on these rather than matching error text.
127
- *
128
- * - `task_terminal` — the task already reached `Complete`/`Failed`/
129
- * `Cancelled`. Checked FIRST, so a terminal task that is also (obviously)
130
- * not `Running` reports the more specific truth, and a steer racing a
131
- * terminal transition always resolves terminal-first.
132
- * - `task_not_running` — the task exists and is live, but is `Offered`/
133
- * `Claimed`/`AwaitApproval`; there is no running turn to steer yet.
134
- * - `steer_unsupported_runtime` — the runtime that CLAIMED this task cannot
135
- * be steered, per the claim-time capability snapshot
136
- * (`TaskSnapshot.claimedRuntimeCapabilities`, sourced from the claiming
137
- * adapter's own `task.claim.capabilities`). Fail-closed: a MISSING snapshot
138
- * rejects under this same code, because "unknown" is not "supported" (see
139
- * {@link SteerRejectedError}'s own doc comment).
140
- */
141
- export type SteerRejectionCode = 'steer_unsupported_runtime' | 'task_not_running' | 'task_terminal';
142
- /**
143
- * S0 (GAP-002): thrown by {@link ConnectionHub.steerTask} instead of the
144
- * pre-S0 generic `Error`, so a caller can tell WHY a steer was refused
145
- * without matching on message text — same typed-error idiom as
146
- * {@link UnknownTaskError}/{@link TaskNotAwaitingApprovalError}/
147
- * {@link StaleApprovalError} above.
148
- *
149
- * The gap this closes: pre-S0 `steerTask` sent `task.steer` to ANY `Running`
150
- * task, but only pi's adapter implements steering — Claude's and Codex's
151
- * throw on receipt (`claude-adapter.ts`, `codex-adapter.ts`), which stalls
152
- * the client's redelivery cursor at that seq and loops forever. So the
153
- * decision has to be made server-side, from per-runtime truth, BEFORE an
154
- * envelope exists.
155
- *
156
- * Fail-closed on unknown, deliberately: `steer_unsupported_runtime` covers
157
- * both "the claiming adapter reported `steer: false`" and "this server has no
158
- * capability snapshot for this task at all" (a pre-D-4 daemon whose claim
159
- * carried no `capabilities`, a task record predating S0). Refusing an unknown
160
- * is a recoverable operator-visible error; guessing "supported" reintroduces
161
- * the exact permanent cursor stall this gate exists to prevent.
162
- *
163
- * Thrown before any state change and before any wire message is built — a
164
- * rejected steer has zero side effects, same as every other validation
165
- * failure in this file.
166
- *
167
- * SINGLE SOURCE, deliberately: the gate reads the claim payload and NOTHING
168
- * from the connection layer — not {@link getDeviceCapabilities} (the
169
- * CONNECTION-level `conn.hello` flag list) and not `ConnectionState.runtimes`.
170
- * Two reasons, either sufficient. First, scope: connection-level data is
171
- * discovery describing a daemon BUILD, not the per-runtime, per-task,
172
- * claim-time truth this gate needs — conflating the two is the original bug.
173
- * Second, lifecycle: the authenticated `conn.hello` snapshot on either
174
- * transport describes the device build, while the claim establishes the
175
- * exact task↔runtime binding. Only the latter shares a lifecycle with the
176
- * thing this gate judges. Adding a connection-level fallback here would
177
- * restore the scope defect and is what this design exists to forbid.
178
- */
179
- export declare class SteerRejectedError extends Error {
180
- readonly taskId: string;
181
- readonly code: SteerRejectionCode;
182
- /** The task's state at the moment the steer was refused. */
183
- readonly state: TaskState;
184
- /** `TaskSnapshot.claimedRuntime` — `undefined` when nothing was ever recorded (which is itself a reason `steer_unsupported_runtime` can fire). */
185
- readonly runtime: RuntimeId | undefined;
186
- constructor(taskId: string, code: SteerRejectionCode,
187
- /** The task's state at the moment the steer was refused. */
188
- state: TaskState,
189
- /** `TaskSnapshot.claimedRuntime` — `undefined` when nothing was ever recorded (which is itself a reason `steer_unsupported_runtime` can fire). */
190
- runtime: RuntimeId | undefined);
191
- }
192
- export type AgentHomeProjectionCompletionErrorCode = 'not_found' | 'conflict' | 'invalid';
193
- export declare class AgentHomeProjectionCompletionError extends Error {
194
- readonly code: AgentHomeProjectionCompletionErrorCode;
195
- constructor(code: AgentHomeProjectionCompletionErrorCode, message: string);
196
- }
197
- /** A requested delivery cursor predates controls this bounded hub can recover. */
198
- export declare class ReplayGapError extends Error {
199
- readonly cursor: number;
200
- readonly recoverableFrom: number;
201
- constructor(cursor: number, recoverableFrom: number);
202
- }
203
- export declare class ConnectionHub {
204
- private readonly taskStore;
205
- private readonly devices;
206
- /** See {@link CreateByokServerOptions.taskLeaseMs} — already defaulted by `createByokServer` before reaching here. */
207
- private readonly taskLeaseMs;
208
- /**
209
- * M4 Phase 4 (part A): per-device inbound-envelope token bucket — see
210
- * {@link CreateByokServerOptions.rateLimit} (already defaulted by
211
- * `createByokServer` before reaching here) and {@link handleInbound}'s
212
- * own doc comment for where it's enforced. Defaults to a fresh
213
- * default-configured `RateLimiter` so every existing direct-construction
214
- * call site (this hub is constructed directly by several tests) keeps
215
- * working unchanged.
216
- */
217
- private readonly rateLimiter;
218
- private readonly agentMessage?;
219
- private readonly connections;
220
- private readonly connectionEpochs;
221
- private readonly outboxes;
222
- /** Idempotency window per device (N3) — recent inbound envelope ids, capped at {@link DEDUP_RING_CAPACITY}. */
223
- private readonly dedupRings;
224
- private readonly longPollWaiters;
225
- private readonly runtimes;
226
- private readonly serverEvents;
227
- /** First-write-wins reliable facts; the reference composition's bounded in-memory readback. */
228
- private readonly agentEgressReceipts;
229
- private readonly agentMessageRequirements;
230
- private readonly agentMessageReceipts;
231
- private readonly agentMessageTaskPayloads;
232
- /** Accepted requests are the authority that later receipts/transfers must echo exactly. */
233
- private readonly agentContentReadRequests;
234
- /** Content-free explicit-read audit facts keyed by exact authenticated device/request identity. */
235
- private readonly agentContentReceipts;
236
- /** Immutable requested projection facts, keyed by authenticated device/request. */
237
- private readonly agentHomeProjectionRequests;
238
- /** First terminal completion for each exact projection request. */
239
- private readonly agentHomeProjectionCompletions;
240
- /**
241
- * Per-task last-inbound-activity timestamp (epoch ms) — the task-lease
242
- * reaper's condition (c), see the "task-lease reaper" section below. Reset
243
- * on every accepted inbound `task.*` envelope ({@link recordTaskActivity},
244
- * called from {@link dispatchToHandler}); cleared once the task reaches a
245
- * terminal state ({@link onStateChange}), so this map only ever holds
246
- * entries for currently non-terminal claimed tasks.
247
- */
248
- private readonly taskActivity;
249
- /** The task-lease reaper's own periodic sweep timer — see the constructor and `sweepLeases` below. */
250
- private readonly leaseReaperTimer;
251
- /** {@link ConnectionHub.stats}'s `uptimeMs` origin — this hub's own construction instant. */
252
- private readonly startedAtMs;
253
- /** {@link ConnectionHub.stats}'s `envelopesIn` — every {@link handleInbound} call, every outcome. */
254
- private envelopesInCount;
255
- /** {@link ConnectionHub.stats}'s `envelopesOut` — every envelope built via the single outbound choke point, {@link sendToDevice}. */
256
- private envelopesOutCount;
257
- /** {@link ConnectionHub.stats}'s `dedupDrops` (N3). */
258
- private dedupDropCount;
259
- /** {@link ConnectionHub.stats}'s `rateLimitEvents` — see {@link handleRateLimited}. */
260
- private rateLimitEventCount;
261
- /**
262
- * M4 Phase 4 (gatekeeper LOW advisory): devices that have already had a
263
- * `device.rate_limited` embedder event emitted for their CURRENT
264
- * over-budget episode — see {@link handleRateLimited}'s own doc comment.
265
- * Coalescing state only; {@link rateLimitEventCount} still counts every
266
- * single hit regardless of what this suppresses.
267
- */
268
- private readonly rateLimitEventEmittedFor;
269
- constructor(taskStore: TaskStore, devices: DeviceRegistry,
270
- /** See {@link CreateByokServerOptions.taskLeaseMs} — already defaulted by `createByokServer` before reaching here. */
271
- taskLeaseMs: number,
272
- /**
273
- * M4 Phase 4 (part A): per-device inbound-envelope token bucket — see
274
- * {@link CreateByokServerOptions.rateLimit} (already defaulted by
275
- * `createByokServer` before reaching here) and {@link handleInbound}'s
276
- * own doc comment for where it's enforced. Defaults to a fresh
277
- * default-configured `RateLimiter` so every existing direct-construction
278
- * call site (this hub is constructed directly by several tests) keeps
279
- * working unchanged.
280
- */
281
- rateLimiter?: RateLimiter, agentMessage?: {
282
- consume(input: {
283
- readonly deviceId: string;
284
- readonly taskId: string;
285
- readonly context: AgentMessageServerContext;
286
- readonly payload: AgentMessagePublishPayload;
287
- }): {
288
- readonly outcome: 'accepted' | 'held' | 'refused';
289
- readonly reasonCode?: string;
290
- };
291
- } | undefined);
292
- /**
293
- * Stop the task-lease reaper's sweep timer — called by `ByokServer.stop()`
294
- * (`index.ts`) on shutdown. Idempotent: clearing an already-cleared
295
- * interval is a safe no-op.
296
- */
297
- stopLeaseReaper(): void;
298
- /** The top-level `events` feed returned by `createByokServer` — see {@link ByokServerEvent}. */
299
- subscribeServerEvents(): AsyncIterable<ByokServerEvent>;
300
- /**
301
- * A daemon completed the WS handshake (`conn.hello`). Does not itself send
302
- * `conn.ack` or redeliver — see {@link sendConnAck}/{@link redeliverAfterReconnect}.
303
- *
304
- * `capabilities` (M5, hello-capability plumbing): the daemon's own
305
- * `conn.hello.capabilities` — previously silently ignored end to end (a
306
- * verified gap: `ws-server.ts` forwarded only `runtimes`). Optional so
307
- * every pre-M5 direct-construction call site (several tests construct a
308
- * `ConnectionHub` and call this directly) keeps working unchanged; a
309
- * connection this hub never learns capabilities for simply reads back
310
- * `undefined` from {@link getDeviceCapabilities}.
311
- */
312
- registerConnection(deviceId: string, ws: WebSocket, runtimes: RuntimeInfo[] | undefined, capabilities?: readonly string[], configuredToolsets?: readonly ToolsetId[], clientVersion?: string): number;
313
- /** True only while this exact WS registration remains the device authority. */
314
- isCurrentConnection(deviceId: string, ws: WebSocket, epoch: number): boolean;
315
- sendConnAck(deviceId: string, capabilities: string[]): void;
316
- /**
317
- * Reconnection procedure step 3 (§9): redeliver, in `seq` order, every
318
- * retained envelope with `seq > cursor` that still belongs to a
319
- * non-terminal task. Called after `conn.ack` (step 2), per the spec.
320
- */
321
- redeliverAfterReconnect(deviceId: string, cursor: number): void;
322
- /**
323
- * A device's WS socket closed. `ws` identifies *which* socket closed: if
324
- * it's no longer the one this device's connection state points at (a
325
- * newer WS reconnected, or long-poll took over — "last transport wins"),
326
- * this close is for a stale/superseded socket and the device isn't
327
- * actually gone, so the bookkeeping below is skipped entirely.
328
- *
329
- * M1 note: the M0 server force-failed/cancelled every in-flight task for a
330
- * device the instant it disconnected, on the stated premise that "a task
331
- * still in flight for a device that just disconnected can't be resumed, so
332
- * it's terminated" — true only in the absence of a redelivery cursor. M1
333
- * adds exactly that (§9): a task's in-flight state is retained
334
- * independently of any one connection, specifically so it can survive a
335
- * disconnect and resume via redelivery once the device reconnects. Failing
336
- * tasks here would make that feature unreachable in practice (nothing
337
- * would ever still be non-terminal by the time a reconnect happened), so
338
- * this now only updates connection bookkeeping and leaves task state
339
- * alone. A task left in-flight by a device that never reconnects stays
340
- * that way until the SaaS embedder explicitly cancels it — no
341
- * disconnect-timeout is specified by the protocol, so none is invented
342
- * here (see the M1-2 report's contract-gap notes).
343
- */
344
- handleDisconnect(deviceId: string, ws: WebSocket): void;
345
- /**
346
- * Drop every piece of hub state keyed by `deviceId` — called when the
347
- * device's registration is DELETED (§6.3 revocation, composed in
348
- * `index.ts`). Presence, its undelivered outbox, its inbound-dedup ring,
349
- * any held long-poll, and its rate-limit episode suppression only ever
350
- * existed to serve a row the directory can no longer name; leaving them
351
- * would keep a deleted device visible as live state and would let a
352
- * re-paired device inherit the dead one's dedup window and seq cursor.
353
- *
354
- * What the device DID is not revocation's business and is untouched here:
355
- * task records (`taskStore`), egress and content receipts, and projection
356
- * facts are history, not credentials.
357
- *
358
- * A live socket is closed rather than left attached — a still-open WS whose
359
- * next envelope would re-create the presence entry we just deleted is a
360
- * half-applied revocation. The close is detached first so `handleDisconnect`
361
- * sees no connection state and skips its disconnect bookkeeping: the device
362
- * is not going dark, it is gone.
363
- */
364
- forgetDevice(deviceId: string): void;
365
- /**
366
- * Resolve immediately if there are already-relevant events past `cursor`;
367
- * otherwise hold for up to `holdMs` and resolve with an empty result if
368
- * nothing arrives. A device may be connected via WS or long-poll, not
369
- * both simultaneously — a poll here supersedes (closes) any live WS for
370
- * this device ("last one wins", documented at the type level on
371
- * {@link ConnectionState}).
372
- */
373
- pollEvents(deviceId: string, cursor: number, holdMs: number): Promise<{
374
- events: Envelope[];
375
- cursor: number;
376
- }>;
377
- /** Make long-poll this device's active transport, closing any live WS ("last one wins", §8). */
378
- private takeOverAsLongPoll;
379
- /** Resolve (settle) any long-poll request currently held open for `deviceId`, if one exists. */
380
- private settleLongPollWaiter;
381
- /**
382
- * Single inbound choke point for every daemon -> server envelope (N2/N3/
383
- * P2) — called by both the WS path (`ws-server.ts`) and the long-poll send
384
- * path (`POST /byok/messages`, `http.ts`) in place of reaching into
385
- * per-type handlers directly. Runs a fixed gate, in order:
386
- *
387
- * 0. **rate limit (M4 Phase 4, part A)** — one token debited from this
388
- * device's bucket ({@link rateLimiter}) for EVERY inbound envelope,
389
- * before anything else runs (including the type-allow check below) —
390
- * a flood of garbage-typed envelopes must cost the same budget as a
391
- * flood of well-formed ones. Checked first specifically so an
392
- * over-budget device is turned away as cheaply as possible, before any
393
- * taskStore lookup or dedup bookkeeping. See {@link handleRateLimited}
394
- * for what happens on exceed (never a silent drop).
395
- * 1. **type-allow (P2)** — only {@link DAEMON_TO_SERVER_TYPES} may pass; a
396
- * server -> daemon type arriving inbound is rejected before it's
397
- * dispatched or counted accepted. `conn.hello` is the one non-task
398
- * exception, and is accepted only from the bearer-authenticated
399
- * long-poll route with an exact device/product/protocol match.
400
- * 2. **ownership (N2)** — an envelope for a task already owned by a
401
- * *different* device is dropped (logged), never force-failed:
402
- * force-failing on an authz mismatch would let an attacker who merely
403
- * guesses a `taskId` kill the real owner's task (a DoS). A task with no
404
- * owner yet, or that doesn't exist at all, is not rejected here — the
405
- * per-type handler's own no-op-on-missing-record behavior covers the
406
- * latter.
407
- * 3. **dedup (N3)** — an envelope `id` already seen from this device is a
408
- * no-op: the wire is at-least-once (§9), this makes server-side
409
- * processing at-most-once. Check-and-record is synchronous (Node is
410
- * single-threaded), so it's atomic with respect to any other envelope
411
- * for this device.
412
- * 4. **dispatch** — handed to the existing per-type `on*` handler.
413
- *
414
- * Returns which outcome applied. A duplicate still counts as `accepted` on
415
- * the `POST /byok/messages` wire (§8.2) — an idempotent replay is a
416
- * wire-level success even though no handler ran a second time; only
417
- * `rejected`/`rate_limited` (gate steps 0-2) are excluded from that count.
418
- */
419
- handleInbound(deviceId: string, envelope: Envelope, authenticatedProductId?: string): 'accepted' | 'duplicate' | 'rejected' | 'rate_limited';
420
- /**
421
- * Store before acking. Replays must agree on every identity/cursor/hash
422
- * field and receive the original receipt id; a same event id with changed
423
- * facts is rejected rather than treated as an update.
424
- */
425
- private handleAgentEgressReliable;
426
- private handleAgentMessagePublish;
427
- private sendAgentMessageDisposition;
428
- private handleAgentContentReceipt;
429
- private sendAgentEgressAck;
430
- private sendAgentContentReceiptAck;
431
- private agentEgressReceiptKey;
432
- /** Record the authenticated long-poll equivalent of the WS opening frame. */
433
- private registerLongPollHello;
434
- /**
435
- * M4 Phase 4 (part A): `deviceId` just exceeded its inbound-envelope rate
436
- * limit. Never a silent drop: counts the occurrence
437
- * ({@link rateLimitEventCount}, surfaced via {@link stats} — every single
438
- * hit, unconditionally) and, the FIRST time in this over-budget episode
439
- * only, emits an embedder-facing `device.rate_limited`
440
- * {@link ByokServerEvent} — see that variant's own doc comment (`types.ts`)
441
- * for the full per-transport enforcement shape.
442
- *
443
- * Gatekeeper LOW advisory (event amplification): a single flood can make
444
- * `handleInbound` call this many times in a row — e.g. several WS frames
445
- * already in flight before the close below actually lands, or a
446
- * long-poll device retrying its `POST /byok/messages` before its bucket
447
- * has refilled. Without coalescing, an embedder subscribed to
448
- * `events.subscribe()` would see one `device.rate_limited` per hit, which
449
- * is noisy for what is really ONE ongoing episode of one device
450
- * flooding. `rateLimitEventEmittedFor` suppresses the repeats: this
451
- * method only pushes the event the first time it sees a given `deviceId`
452
- * since `handleInbound`'s own success path last cleared it (i.e. since
453
- * this device was last confirmed back under budget) — the COUNTER above
454
- * is entirely unaffected by this and still increments on every call,
455
- * unconditionally.
456
- *
457
- * This method only handles the WS half of the enforcement shape (closing
458
- * the live connection, if any, so the client's existing backoff+reconnect
459
- * takes over — mirrors `takeOverAsLongPoll`'s own `ws.close`, the only
460
- * other place this hub closes a device's socket directly); a long-poll
461
- * device has no live `ws` to close here at all (`conn.ws` is `undefined`
462
- * while long-polling — see {@link ConnectionState}), so `http.ts`'s
463
- * `/byok/messages` handler maps this same `'rate_limited'` `handleInbound`
464
- * outcome to an HTTP 429 for that transport instead.
465
- */
466
- private handleRateLimited;
467
- /**
468
- * Idempotency check-and-record (N3): `true` (duplicate) if `id` was
469
- * already seen for `deviceId`; otherwise records it and returns `false`.
470
- * Bounded to {@link DEDUP_RING_CAPACITY} ids per device — a ring, not an
471
- * unbounded set — evicting the oldest once full.
472
- */
473
- private checkAndRecordDuplicate;
474
- /**
475
- * Route one already-gated envelope (see {@link handleInbound}) to its
476
- * per-type handler. Type-allow/ownership/dedup have already run by the
477
- * time this executes, so the handlers below no longer need their own
478
- * device-mismatch checks — that authz decision now lives solely in
479
- * `handleInbound` (N2).
480
- *
481
- * Also the task-lease reaper's activity checkpoint
482
- * ({@link recordTaskActivity}): every envelope for a task that currently
483
- * *exists and is non-terminal* counts as proof of life for `taskId`'s
484
- * lease, regardless of what its per-type handler below ends up doing with
485
- * it (including a no-op/stale drop) — see the "task-lease reaper" section
486
- * further down for why. Deliberately gated on the record's existence and
487
- * non-terminal state *here*, before dispatch: `taskActivity` must never
488
- * gain an entry for a taskId that doesn't exist (a nonexistent/garbage id
489
- * an authenticated-but-malicious daemon could send indefinitely — an
490
- * unbounded-growth vector, since `taskId`s aren't deduped the way envelope
491
- * `id`s are) or for one that's already terminal (a stale/late message for
492
- * a finished task — `onStateChange` deletes the entry on the *real*
493
- * terminal transition, but a stale message arriving *after* that would
494
- * otherwise silently recreate it, since every per-type handler's own
495
- * terminal/unknown-task guard runs — and early-returns — only *after*
496
- * this would already have recorded activity).
497
- */
498
- private dispatchToHandler;
499
- /** Reset the task-lease reaper's per-task clock (condition (c) in the "task-lease reaper" section below). */
500
- private recordTaskActivity;
501
- /**
502
- * Ownership (record.deviceId matching the connection's authenticated
503
- * deviceId) is enforced centrally by {@link handleInbound} (N2) before this
504
- * runs; only the idempotent-claim CAS and the first-claim device patch
505
- * happen here.
506
- *
507
- * M5 (claimed runtime, docs/protocol.md §3.1): `payload.runtime` — the
508
- * ACTUAL adapter the daemon selected (`TaskRunner.pickAdapter`,
509
- * `packages/client`'s `task-runner.ts`) — is recorded into
510
- * `TaskSnapshot.claimedRuntime` alongside the device patch, distinct from
511
- * the pre-existing `TaskSnapshot.runtime` (the merely REQUESTED runtime,
512
- * untouched here and set only once, at `dispatch()` time). Only ever
513
- * written on the FIRST real claim: the idempotent-CAS early return above
514
- * fires before this for a retried claim from a device that already owns
515
- * the task, so a redelivered/retried `task.claim` can never overwrite an
516
- * already-recorded `claimedRuntime` — including with a stale or absent
517
- * value from an out-of-order retry.
518
- *
519
- * S0/D-4 (claim-time capability snapshot): `payload.capabilities` — the
520
- * claiming adapter's OWN self-report, carried on this same `task.claim`
521
- * (docs/protocol.md §2.4) — supplies
522
- * `TaskSnapshot.claimedRuntimeCapabilities`, written in the same patch and
523
- * therefore under the same write-exactly-once property as `claimedRuntime`.
524
- *
525
- * Taken from the payload and from nowhere else. This hub deliberately does
526
- * NOT consult connection state (`conn.hello.runtimes[]`) for it — see
527
- * {@link SteerRejectedError} for why that source is structurally wrong for a
528
- * control decision, and that field's own doc comment (`types.ts`) for why
529
- * this is snapshotted rather than read live at steer time. A claim that
530
- * carries no `capabilities` (a pre-D-4 daemon) records `undefined`, which
531
- * the gate reads as "unknown" and refuses.
532
- */
533
- private onClaim;
534
- /**
535
- * `Claimed -> Running` (§3.1) — a daemon actually starting the runtime
536
- * session, distinct from merely claiming. Ownership is already enforced
537
- * by {@link handleInbound} (N2) before this runs.
538
- */
539
- private onStarted;
540
- /**
541
- * `Offered -> Failed` (§3.2) — a fail-closed pre-claim rejection. Only
542
- * ever legal from `Offered`; anything else is stale. Ownership is already
543
- * enforced by {@link handleInbound} (N2) before this runs.
544
- */
545
- private onDecline;
546
- private onProgress;
547
- private onArtifact;
548
- private onAwaitApproval;
549
- private onComplete;
550
- private onFail;
551
- /**
552
- * Dual-purpose on receipt (§3.3): if the server already moved this task to
553
- * `Cancelled` on its own action (the common case — `cancelTask()` is
554
- * authoritative immediately, §4), this is a late idempotent ack — silent,
555
- * not a warning (this is the other half of the M0 gatekeeper finding this
556
- * change resolves). Otherwise it's the authoritative trigger for a
557
- * cancellation the daemon observed that the server didn't initiate.
558
- * Ownership is already enforced by {@link handleInbound} (N2) before this
559
- * runs.
560
- */
561
- private onCancelled;
562
- /**
563
- * M4 (additive-minor, `task.approval_resolved`): the EXPLICIT counterpart
564
- * to {@link resumeIfImplicitlyApproved} — a daemon that resolved a pending
565
- * approval entirely LOCALLY now reports it immediately, instead of the
566
- * server only finding out after the fact once evidence (a later
567
- * `task.progress`/`task.artifact`/`task.complete`) proves it.
568
- *
569
- * Relationship to the implicit path (both stay, permanently — this is not
570
- * a replacement): {@link resumeIfImplicitlyApproved} remains completely
571
- * untouched as the fallback for (a) an old daemon that predates this
572
- * message, and (b) a daemon connected to an old server that never
573
- * advertised the `approval_resolved` capability flag (`version.ts`) at
574
- * handshake time — in either case the daemon never sends this message at
575
- * all (see `packages/client`'s `task-runner.ts`), and the server keeps
576
- * inferring the resolution from evidence exactly as it did before this
577
- * message existed. When THIS message does arrive first, it already moves
578
- * the record out of `AwaitApproval` (see below) — so by the time any
579
- * following `task.progress`/etc. reaches `onProgress`/`onArtifact`/
580
- * `onComplete`, `resumeIfImplicitlyApproved`'s own `record.state !==
581
- * 'AwaitApproval'` guard is already true and it no-ops, never firing its
582
- * own `task.approval_resolved_implicit` event a second time for the same
583
- * resolution. The two mechanisms race harmlessly: whichever one the
584
- * server processes first is the one that actually performs the
585
- * transition; the other is naturally inert once it runs.
586
- *
587
- * Three outcomes, mirroring this file's existing per-type idempotency
588
- * conventions:
589
- * - `AwaitApproval` (the expected case): legal transition to `Running`
590
- * (an existing `TASK_TRANSITIONS` edge, the same one `approveTask`
591
- * itself uses) plus a `task.approval_resolved` {@link ByokServerEvent}
592
- * carrying `approvalId`/`decision`/`resolvedBy` for an embedder to
593
- * observe.
594
- * - Already `Running` (evidence — or the implicit path — already beat
595
- * this message to it): idempotent no-op, silent, mirroring
596
- * `onStarted`'s own already-running guard.
597
- * - Terminal, or a state that was never `AwaitApproval` in the first
598
- * place (`Offered`/`Claimed` — a genuinely out-of-sequence report):
599
- * stale no-op with a `console.warn`, matching this file's existing
600
- * stale-message convention (`forceFailOrDrop`, `handleInbound`'s
601
- * ownership-mismatch drop) — never force-failed, since a late/
602
- * redelivered report about a task that has already moved on is not
603
- * evidence of anything currently wrong with it.
604
- *
605
- * This is also the residual-race resolution the accompanying protocol/docs
606
- * update documents: a SaaS decision (`approveTask`/`rejectTask`) already in
607
- * flight when the local resolution happens can still land on the server
608
- * FIRST and move the record to a terminal state before this message
609
- * arrives — in that case this message hits the terminal branch above and
610
- * is a stale no-op, exactly like any other late message for an
611
- * already-terminal task. The window for that crossing is now
612
- * network-latency-sized (how long this message takes to arrive), not
613
- * "until the next progress message" the way the pre-existing implicit-only
614
- * inference left it.
615
- */
616
- private onApprovalResolved;
617
- /**
618
- * M5 (approval targeting): single low-level wrapper around
619
- * `TaskStore.transition` that every ACTUAL state-changing write in this
620
- * file goes through — {@link applyOrFail}'s legal-transition branch,
621
- * {@link forceFailOrDrop}, and {@link resumeIfImplicitlyApproved} (the one
622
- * caller that transitions WITHOUT going through `applyOrFail` at all).
623
- * Two responsibilities, folded in here once rather than duplicated at
624
- * each call site:
625
- *
626
- * 1. Clears `pendingApprovalId` whenever `record` is LEAVING
627
- * `AwaitApproval` (`record.state === 'AwaitApproval' && to !==
628
- * 'AwaitApproval'`) — the id this hub last recorded for a task's
629
- * pending approval ({@link onAwaitApproval}) is meaningless the
630
- * instant that task is no longer awaiting it. Clearing it here,
631
- * centrally, is what guarantees a FUTURE `AwaitApproval` cycle for
632
- * the SAME task always starts from a clean slate instead of silently
633
- * inheriting a stale id from a previous cycle (which would make a
634
- * stale-approval check against the NEW cycle's real pending id
635
- * spuriously pass just because a leftover value happened to still be
636
- * sitting in the record).
637
- * 2. Calls {@link onStateChange} — every call site already did this
638
- * immediately after its own `transition` call; folding it in here
639
- * removes the duplication and the chance of a future call site
640
- * forgetting it.
641
- */
642
- private transitionTask;
643
- /**
644
- * Apply `taskId`'s state -> `target`. If that's illegal per
645
- * `TASK_TRANSITIONS`, fall back to `Failed` (if reachable from the current
646
- * state); this is the "illegal transition = error + task.fail path" rule.
647
- */
648
- private applyOrFail;
649
- /**
650
- * M4 Phase 3 hardening (orchestrator-directed fix for the server-state-
651
- * machine trace finding): a task can be resolved entirely OUT-OF-BAND, on
652
- * the daemon side only (M4 Phase 3's local `approvals.resolve`
653
- * control-socket path, `packages/client`) — the server never sees a wire
654
- * `task.approve`/`task.reject` for it, so its own record sits in
655
- * `AwaitApproval` even though the daemon already resumed and moved on.
656
- *
657
- * The daemon is the execution authority in this security model (the SaaS
658
- * only ever *proposes* — see docs/spec.md); the daemon sending ANY further
659
- * task.* traffic for a task the server still thinks is `AwaitApproval` is
660
- * itself sufficient proof the approval was resolved locally, one way or
661
- * another. Rather than force-failing/dropping that traffic (the pre-fix
662
- * behavior — `onProgress`/`onArtifact`'s own `!== 'Running'` guard,
663
- * `onComplete`'s illegal-transition fallback), this applies the exact same
664
- * `AwaitApproval -> Running` edge `approveTask` already uses (a
665
- * pre-existing legal `TASK_TRANSITIONS` edge, not a new one) through the
666
- * normal transition path — `taskStore.transition` + `onStateChange`, same
667
- * as `applyOrFail`'s own legal-transition branch — so every existing
668
- * consumer of task state (§, `TaskHandle.events()`, the lease reaper's
669
- * `taskActivity`) observes it exactly as it would a real wire
670
- * `task.approve`. Then emits `task.approval_resolved_implicit` (a
671
- * `ByokServerEvent`, NOT a wire message — see that type's own doc comment)
672
- * so an embedder can distinguish this from an operator-driven approval.
673
- *
674
- * M4 (additive-minor, superseding this method's own former "deferred"
675
- * framing): a first-class `task.approval_resolved` WIRE notification now
676
- * exists (`onApprovalResolved`, below) — a daemon that supports it, talking
677
- * to a server that advertised the `approval_resolved` capability flag
678
- * (`version.ts`), reports a local resolution explicitly and immediately
679
- * instead of leaving the server to infer it here. This method is
680
- * UNTOUCHED and remains the permanent fallback for the N/N-1 cases where
681
- * that explicit report never arrives (an old daemon, or an old server this
682
- * daemon is talking to) — see `onApprovalResolved`'s own doc comment for
683
- * the full relationship between the two paths, including why they can
684
- * never both fire for the same resolution.
685
- *
686
- * No-op (returns `record` unchanged) for any state other than
687
- * `AwaitApproval` — every other guard (terminal, pre-claim, already-
688
- * Running) keeps exactly its current behavior. `onFail`/`onCancelled`
689
- * never call this: `Failed`/`Cancelled` are already direct, legal edges
690
- * from `AwaitApproval`, so they never hit the illegal-transition path this
691
- * exists to avoid in the first place.
692
- */
693
- private resumeIfImplicitlyApproved;
694
- /**
695
- * A daemon message didn't fit the task's current state (e.g. progress
696
- * while AwaitApproval). Force the task to `Failed` if that's reachable;
697
- * otherwise it's already terminal (or `Offered`, which has no Failed edge)
698
- * and there's nothing safe to do but log + drop.
699
- */
700
- private forceFailOrDrop;
701
- private onStateChange;
702
- /**
703
- * Task lease: a backstop for a device that goes dark mid-task and never
704
- * comes back — distinct from, and layered on top of, M1's redelivery
705
- * (docs/protocol.md §9), which already handles "device reconnects within
706
- * the window, nothing lost." Decision (user+design): reuse the existing
707
- * `Failed` terminal state and its `retryable` flag —
708
- * `Failed(retryable: true, reason: 'lease-expired')` — exactly like any
709
- * other `task.fail`. The embedder is expected to treat this exactly like
710
- * any other retryable failure: re-dispatch as a brand-new task.
711
- *
712
- * Implemented as a periodic sweep (see the constructor), not a per-task
713
- * timer, so a device that goes dark *after* being idle-but-connected for a
714
- * while is still caught on a later tick without needing extra bookkeeping
715
- * at disconnect time. `sweepLeases` reaps a task only when ALL of the
716
- * following hold, checked fresh on every tick (never cached):
717
- *
718
- * (a) the task is in a non-terminal *claimed* state — `Claimed`,
719
- * `Running`, or `AwaitApproval` ({@link isClaimedState}). `Offered`
720
- * is excluded: it has no owning device yet, so there's nothing to
721
- * be "dark".
722
- * (b) the owning device is dark right now ({@link deviceDarkSince}
723
- * returns a timestamp rather than `undefined`) — disconnected
724
- * outright, or (long-poll only) hasn't been seen since before the
725
- * lease window. A live WS connection is never dark from the
726
- * reaper's point of view: `heartbeat.ts` already independently
727
- * proves liveness at the transport level and flips
728
- * `connected: false` via `handleDisconnect` once it stops getting
729
- * pongs — the reaper just reads that flag rather than re-deriving
730
- * it. `deviceDarkSince` also returns *when* darkness started
731
- * ({@link ConnectionState.darkSince}, set the instant
732
- * `handleDisconnect` flips the connection dark) — that instant
733
- * feeds condition (c), below.
734
- * (c) a full `taskLeaseMs` has elapsed since the *later* of: the task's
735
- * own last inbound-activity timestamp ({@link taskActivity}, reset
736
- * in {@link dispatchToHandler} on every accepted envelope for a
737
- * known, non-terminal task — claim, started, progress, artifact,
738
- * await_approval, anything), and (b)'s dark-since instant. Taking
739
- * the *later* of the two — not the activity timestamp alone — is
740
- * what makes a device going dark start a fresh, full countdown
741
- * instead of reusing whatever (possibly already-stale) activity
742
- * timestamp the task happened to have: a task can be legitimately
743
- * idle *while connected* for longer than `taskLeaseMs` (a long turn
744
- * with no progress events, or just a quiet stretch) without being
745
- * touched — see (b) — but the instant such a task's device
746
- * disconnects, that stale activity timestamp must NOT immediately
747
- * satisfy (c) on its own, or the task would get reaped within one
748
- * sweep tick of disconnect instead of waiting the full window. That
749
- * was a real bug (a disconnect-after-long-idle reap effectively
750
- * indistinguishable from the M0 disconnect-alone-fails-the-task
751
- * behavior M1 removed, below); anchoring (c) to
752
- * `max(lastActivity, darkSince)` fixes it — idle time that elapsed
753
- * *before* the device went dark no longer counts toward the lease,
754
- * only silence *after* dark-start does.
755
- *
756
- * (b) and (c) are deliberately independent clocks, not one merged check.
757
- * The property this most exists to protect: a *connected*, momentarily
758
- * idle device mid-turn must never be reaped, no matter how long
759
- * `taskLeaseMs` is — condition (b) alone blocks that regardless of (c).
760
- * This is also what keeps this from reintroducing the M0 bug M1
761
- * deliberately removed (see `handleDisconnect`'s own doc comment above) —
762
- * M0 force-failed a task the instant its device disconnected; M1
763
- * correctly stopped doing that so a task could survive a disconnect and
764
- * resume via redelivery. This reaper does not revert that: disconnect
765
- * ALONE still does nothing here either — (c) still has to independently
766
- * hold, and per the `max(...)` above it only will once a full
767
- * `taskLeaseMs` has genuinely elapsed *since the device went dark*, no
768
- * matter how stale the task's own activity timestamp already was at that
769
- * moment.
770
- *
771
- * Interaction with redelivery (§9): redelivery is what handles "the
772
- * device came back within the window" — nothing to reap, normal traffic
773
- * resumes. This reaper is what handles "it never came back." Idempotent
774
- * claim (`onClaim`'s CAS) still protects server-side bookkeeping if a
775
- * device wakes up *after* its task was already reaped and retries a stale
776
- * claim/progress/etc. for it: every per-type handler's existing
777
- * stale/terminal-task guard (§9) drops it as a no-op, same as any other
778
- * late message for an already-terminal task — no new guard was needed for
779
- * that here.
780
- *
781
- * Accepted residual (by design, not a bug): idempotent claim protects
782
- * *server-side* state, not the device's own local side effects. A dark
783
- * device that wakes up after its task has already been reaped may still
784
- * be mid-way through running real local work (file writes, shell
785
- * commands, whatever the runtime adapter was doing) for a task the server
786
- * has since moved on from — and that the embedder may have already
787
- * re-dispatched elsewhere. There is no way to remotely guarantee a
788
- * truly-dark device stops running; the mitigation is entirely
789
- * `taskLeaseMs` being set far larger than any realistic task duration, so
790
- * this can only happen to a device that was genuinely gone for a very
791
- * long time, not a normal slow turn.
792
- */
793
- private sweepLeases;
794
- /**
795
- * Condition (b) above: `undefined` while `deviceId`'s connection counts as
796
- * alive (never reapable, no matter how stale (c) is); otherwise the
797
- * epoch-ms instant it began counting as "dark" for lease purposes.
798
- * `sweepLeases` combines this with (c)'s own last-activity instant via
799
- * `max(...)` so the full `taskLeaseMs` silence window is always measured
800
- * from whichever of the two happened later.
801
- */
802
- private deviceDarkSince;
803
- /** Reap one lease-expired task through the exact same TaskStore/canTransition path — and terminal-event emission — as any other `task.fail` (see {@link applyOrFail}). */
804
- private reapTask;
805
- dispatch(input: DispatchInput): Promise<TaskHandle>;
806
- dispatchFreshAgentEgress(input: FreshAgentEgressDispatchInput): Promise<TaskHandle>;
807
- private dispatchInternal;
808
- /** Capability-gated control-plane read request; no request enters the outbox on omission. */
809
- requestAgentContentRead(input: AgentContentReadRequest): Promise<void>;
810
- /** Task-free exact-device projection; no task record, runtime or session is created. */
811
- enqueueAgentHomeProjection(input: AgentHomeProjectionRequest): Promise<AgentHomeProjectionReadback>;
812
- readAgentHomeProjection(deviceId: string, requestId: string): AgentHomeProjectionReadback | undefined;
813
- completeAgentHomeProjection(deviceId: string, input: AgentHomeProjectionCompletionRequest): AgentHomeProjectionReadback;
814
- private buildTaskHandle;
815
- /** Idempotent: cancelling an already-terminal task is a no-op, not an error. */
816
- private cancelTask;
817
- /**
818
- * M4 Phase 3: made public (was private through M3) so an embedder can call
819
- * it directly from its own operator-facing surface — there is no
820
- * bearer-authed HTTP route for this on `http.ts`'s own app (see
821
- * `UnknownTaskError`'s own doc comment for why, and
822
- * `examples/basic/server.ts`'s `/api/tasks/:taskId/approve` for the
823
- * intended shape of that embedder-built surface). See this file's own
824
- * `UnknownTaskError`/`TaskNotAwaitingApprovalError` doc comments for why
825
- * the two failure modes are now distinct typed errors rather than a
826
- * single generic `Error`. Every thrown message's TEXT is byte-for-byte
827
- * unchanged from M2/M3 — only the error's type changed (this is still also
828
- * reachable via `TaskHandle.approve()`, unaffected).
829
- */
830
- /**
831
- * M5 (approval targeting, docs/protocol.md §5.3): `opts.approvalId`
832
- * targets a SPECIFIC pending approval rather than "whichever one is
833
- * currently pending" (the pre-M5 default, unchanged when `opts` is
834
- * omitted). Validated FIRST, before any state change or wire send: if
835
- * `opts.approvalId` is supplied and this hub has a recorded
836
- * `pendingApprovalId` for `taskId` that DIFFERS, throws
837
- * {@link StaleApprovalError} — no transition, no `task.approve` sent. If
838
- * this hub never recorded a `pendingApprovalId` (a legacy daemon that
839
- * never reported one), the call proceeds untargeted exactly as before.
840
- * The outgoing `task.approve` carries `approvalId`: the caller-supplied
841
- * one if given, else this hub's own recorded one, else omitted entirely
842
- * (legacy wire shape) — so the daemon can apply its own exact-match check
843
- * whenever this server has an id to offer at all.
844
- */
845
- approveTask(taskId: string, opts?: {
846
- approvalId?: string;
847
- }): Promise<void>;
848
- /**
849
- * M4 Phase 3: made public — see {@link ConnectionHub.approveTask}'s own
850
- * doc comment for the full rationale (identical reasoning applies here).
851
- * M5: same `opts.approvalId` targeting semantics as `approveTask` above —
852
- * see that method's own doc comment.
853
- */
854
- rejectTask(taskId: string, reason?: string, opts?: {
855
- approvalId?: string;
856
- }): Promise<void>;
857
- /**
858
- * S0 (GAP-002): a task-level gate, evaluated in full before any envelope is
859
- * built — see {@link SteerRejectedError} for the gap this closes and why an
860
- * unknown capability must refuse rather than proceed. Order matters:
861
- *
862
- * 1. unknown task — unchanged pre-S0 `Error` (this is not a steer-policy
863
- * decision, and `TaskHandle.steer` can only be reached with a taskId
864
- * this hub minted, so it's a programming error, not an operator one);
865
- * 2. terminal (`Complete`/`Failed`/`Cancelled`) -> `task_terminal`,
866
- * checked BEFORE the `Running` check so a steer racing a terminal
867
- * transition always resolves terminal-first;
868
- * 3. not `Running` (`Offered`/`Claimed`/`AwaitApproval`) ->
869
- * `task_not_running`;
870
- * 4. the claim-time snapshot does not positively say `steer: true` ->
871
- * `steer_unsupported_runtime`, including when there is no snapshot at
872
- * all (fail-closed);
873
- * 5. only then, the pre-existing device-liveness check and the send.
874
- *
875
- * Step 4 reads `TaskSnapshot.claimedRuntimeCapabilities` — the per-runtime,
876
- * per-task value frozen at claim time from the claiming adapter's own
877
- * `task.claim.capabilities` — and reads NO connection state whatsoever:
878
- * not {@link getDeviceCapabilities}, not `ConnectionState.runtimes`, and
879
- * with no fallback to either when the snapshot is absent. See
880
- * {@link SteerRejectedError} for why a connection-sourced input is wrong
881
- * in scope (it describes a daemon build, not this task's runtime).
882
- */
883
- private steerTask;
884
- private pickFirstConnectedDevice;
885
- /**
886
- * Build a server -> daemon envelope with a fresh per-device `seq`, retain
887
- * it in that device's outbox ring buffer, and deliver it now if a live
888
- * transport is available (WS send, or wake a pending long-poll).
889
- *
890
- * `opts`'s type mirrors `createEnvelope`'s own per-type conditional
891
- * requiredness (finding F1) minus `seq` (computed fresh right here on
892
- * every call, never caller-supplied) — so every one of this method's 6
893
- * callers below must supply `taskId` for the 5 types that need it
894
- * (everything except `conn.ack`), same as calling `createEnvelope`
895
- * directly would require.
896
- */
897
- private sendToDevice;
898
- private deliverToDevice;
899
- /**
900
- * Retained envelopes for `deviceId` with `seq > cursor` that still belong
901
- * to a non-terminal task — OR are explicitly exempted from that filter
902
- * (`redeliverThroughTerminal`, N1/F4: `task.cancel`/`task.reject`) — in
903
- * `seq` order. The `seq > cursor` bound is what naturally stops an
904
- * exempted entry from redelivering forever: once the daemon acks it (its
905
- * reported cursor advances past that `seq`), it no longer qualifies here
906
- * on any future reconnect/poll.
907
- */
908
- private collectRelevant;
909
- /**
910
- * A caller at `recoverableFrom - 1` can still receive the first retained
911
- * control. Anything earlier would be a partial tail and must fail closed.
912
- */
913
- assertReplayAvailable(deviceId: string, cursor: number, pendingOutbound?: number): void;
914
- private isRecoverableEntry;
915
- private isTaskTerminal;
916
- /** The highest `seq` assigned to `deviceId` so far — the redelivery cursor to hand back on a poll/reconnect. */
917
- private currentCursor;
918
- private getOrCreateOutbox;
919
- private nextConnectionEpoch;
920
- listMachines(): MachineInfo[];
921
- /**
922
- * M5 (approval targeting, hello-capability plumbing): the capability flags
923
- * `deviceId`'s CURRENT connection advertised in its `conn.hello` —
924
- * `undefined` if this hub has no connection state for the device at all,
925
- * or one that never had capabilities recorded (a pre-M5 daemon, or a
926
- * device this hub only ever saw over long-poll with no prior WS hello —
927
- * see `ConnectionState.capabilities`'s own doc comment). Read fresh from
928
- * live connection state, mirroring `listMachines()`'s own convention; an
929
- * embedder can use this to distinguish a targeting-capable device from a
930
- * legacy one for its own observability/UI purposes (see `version.ts`'s
931
- * `approval-targeting` flag doc comment for why this is informational
932
- * only, never a correctness gate).
933
- */
934
- getDeviceCapabilities(deviceId: string): readonly string[] | undefined;
935
- private hasDeviceCapabilities;
936
- getAgentEgressReceipt(deviceId: string, eventId: string): AgentEgressReceipt | undefined;
937
- getTask(taskId: string): TaskSnapshot | undefined;
938
- listTasks(): TaskSnapshot[];
939
- /**
940
- * A plain, serializable snapshot of this hub's current state, derived from
941
- * existing structures (`connections`, `taskStore`) plus the small counters
942
- * this file already maintains for exactly this purpose — no new
943
- * bookkeeping structures beyond those counters. See {@link HubStats}
944
- * (`types.ts`) for the full field-by-field contract.
945
- */
946
- stats(): HubStats;
947
- }