@byollm/relay 0.1.0-alpha.7 → 0.1.0-alpha.71

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,655 @@
1
+ import { PublicIdentity, CapabilityMatrix, WithheldKind, JobStub, SealedEnvelope, ClaimedStub } from '@byollm/protocol';
2
+
3
+ /**
4
+ * The relay's routing state — byollm_009 §7, reachable at last.
5
+ *
6
+ * §7 described a state machine the direct plane could not produce. There, the
7
+ * site and the upstream are the same party: it seals when it likes, and a job
8
+ * is never claimed-but-unsealed. Here they are different parties, and the gap
9
+ * between them is a state:
10
+ *
11
+ * ```
12
+ * queued ──claim──▶ awaiting-payload ──sealed──▶ ready ──fetch──▶ running
13
+ * ▲ │ │
14
+ * └────────────────────┘ ▼
15
+ * site never seals, or seals too late ok | error | canceled
16
+ * ```
17
+ *
18
+ * The relay cannot seal, so it cannot shortcut this. A payload is encrypted
19
+ * to *the device that claimed it*, and nobody knows which device that is until
20
+ * the claim happens — which is precisely why claim-then-fetch makes a blind
21
+ * relay possible at all. The window is the price.
22
+ *
23
+ * ## What the relay holds, and what it cannot
24
+ *
25
+ * Stubs (metadata the site chose to publish), sealed envelopes it cannot open,
26
+ * and public keys. There is no field on any type in this file that could hold
27
+ * a private key or a plaintext, which is `RELAY_BLIND` expressed as a data
28
+ * model rather than as a policy.
29
+ */
30
+ /** Where a routed job is. */
31
+ type RoutedState = "queued" | "awaiting-payload" | "ready" | "running" | "done";
32
+ /**
33
+ * How long a site has to seal after one of its jobs is claimed.
34
+ *
35
+ * **Distinct from the lease, and distinct from the job's TTL** — byollm_009
36
+ * §7.1. Three clocks, three different questions:
37
+ *
38
+ * - the **TTL** asks how long the work is worth doing at all;
39
+ * - the **lease** asks how long this device gets to run it;
40
+ * - this asks how long we wait for a site that has gone away.
41
+ *
42
+ * Collapsing any pair of them looks harmless until a site restarts during a
43
+ * deploy: with only a lease, the device sits politely holding a job whose
44
+ * payload will never arrive, and the lease's whole minute is spent waiting on
45
+ * a party that is not coming back. Short, because a site that is up answers in
46
+ * milliseconds and a site that is down will not answer sooner for waiting.
47
+ */
48
+ declare const AWAITING_PAYLOAD_MS = 10000;
49
+ /** A job the relay is routing. Metadata and ciphertext, nothing else. */
50
+ /** Why a daemon gave a job back. Only `refused` means "not me, ever". */
51
+ type ReleaseReason = "shutdown" | "pause" | "revoked" | "backend-down" | "refused";
52
+ interface RoutedJob {
53
+ readonly id: string;
54
+ /** Which site enqueued it — the party that will be asked to seal. */
55
+ readonly siteId: string;
56
+ /**
57
+ * Everything the relay knows about the work, which is everything the site
58
+ * chose to publish and not one field more (byollm_009 §6).
59
+ */
60
+ readonly stub: JobStub;
61
+ state: RoutedState;
62
+ /** Set from the claim; the site seals to these keys. */
63
+ claimedBy?: {
64
+ readonly runnerId: string;
65
+ readonly owner: string;
66
+ readonly device: PublicIdentity;
67
+ readonly leaseId: string;
68
+ readonly leaseExpiresAt: number;
69
+ };
70
+ /** When {@link AWAITING_PAYLOAD_MS} runs out for this claim. */
71
+ awaitingUntil?: number;
72
+ /**
73
+ * Runners that released this job with reason `refused` — cloud_008 §2.1.
74
+ *
75
+ * `REFUSAL_NOT_REOFFERED`, which the relay did not implement: it dropped
76
+ * `ReleaseRequest.reason` on the floor. The field's own docstring says why
77
+ * that is not cosmetic — an upstream cannot evaluate a daemon's *local*
78
+ * `named` allowlist, so it may legitimately offer work the daemon then
79
+ * declines, and without a record the two spin between claim and release
80
+ * forever. The direct plane has always kept this list.
81
+ */
82
+ refusedBy: string[];
83
+ /**
84
+ * Runners that may not be offered this job again *yet*, and from when.
85
+ *
86
+ * The middle ground {@link RoutingStore.releaseLeases} had no way to say.
87
+ * `refusedBy` is forever and a bare release is immediate; a control plane
88
+ * declining a job for a reason the world can change — an unfilled mapping
89
+ * slot, a resolution that named another machine, a store that was briefly
90
+ * unreachable — means neither. It means "ask again later", and later needs
91
+ * a number.
92
+ *
93
+ * Keyed by runner because it is a fact about a pairing, not about the job:
94
+ * the same job goes to another device immediately, which is the whole
95
+ * point of not marking it refused.
96
+ */
97
+ retryAfter?: Record<string, number>;
98
+ /**
99
+ * The site withdrew this job — cloud_008 §2.2.
100
+ *
101
+ * A flag rather than a state, because a cancelled job that a device is
102
+ * *running* is not finished: the daemon has to be told, abort its backend
103
+ * call and report `canceled`, and the ordinary `complete` path then closes
104
+ * it. Making it a state would strand the in-flight case between two
105
+ * machines' ideas of what happened.
106
+ */
107
+ cancelled?: boolean;
108
+ /** Sealed to the claiming device by the site. Opaque here. */
109
+ payload?: SealedEnvelope;
110
+ /** Sealed to the site by the device. Opaque here. */
111
+ result?: SealedEnvelope;
112
+ /**
113
+ * The result's clear-text discriminator — byollm_009 §6.1.
114
+ *
115
+ * The one outcome fact the relay is given, and the reason it is given:
116
+ * without it the relay cannot stop dispatching a finished job. A routing
117
+ * hint and never a fact — the *site* verifies it against the sealed
118
+ * outcome, because only the site can open the envelope. The relay acts on
119
+ * it and is entitled to be wrong; a lying daemon costs it a dispatch
120
+ * decision, not a security property.
121
+ */
122
+ disposition?: "ok" | "error" | "canceled";
123
+ }
124
+ /** A device the relay has seen recently. */
125
+ interface Presence {
126
+ readonly runnerId: string;
127
+ readonly owner: string;
128
+ readonly device: PublicIdentity;
129
+ lastSeenAt: number;
130
+ /**
131
+ * What this machine last said it can run — cloud_009, 2026-08-24.
132
+ *
133
+ * **Capabilities are presence data.** They arrive on the same heartbeat as
134
+ * everything else here, they go stale at the same moment and for the same
135
+ * reason, and a machine that stops heartbeating has not stopped being able
136
+ * to run Llama — it has stopped being somewhere we can ask. Keeping them
137
+ * anywhere else would put capability truth outside the interface that owns
138
+ * presence, and every next consumer (a member's usable set, a dashboard's
139
+ * pulse, a degraded-state banner) would re-derive it from a different
140
+ * place. That is the two-owners shape this codebase keeps paying for.
141
+ *
142
+ * Empty is a real answer, not a missing one: it is a paired machine with no
143
+ * healthy backend, which is a legal state the whole connect-first ruling
144
+ * exists to make visible rather than refuse.
145
+ */
146
+ capabilities: CapabilityMatrix;
147
+ /**
148
+ * Kinds this machine is deliberately *not* advertising — byollm_016.
149
+ *
150
+ * Presence data for the same reason capabilities are, and stored beside
151
+ * them rather than derived: two services answering one kind with no
152
+ * `defaults` entry is a state only the daemon can see, and the hub cannot
153
+ * reconstruct it from the matrix — an absent kind and a withheld kind look
154
+ * identical there. Without this field the owner's page can say "nothing
155
+ * serves llm.generate", which is true and useless, instead of "two services
156
+ * answer it and you have not chosen", which is the sentence that ends with
157
+ * the owner doing something.
158
+ *
159
+ * Empty is the normal answer, and means every kind resolved.
160
+ */
161
+ withheld: readonly WithheldKind[];
162
+ }
163
+ /**
164
+ * What a routing store must do, expressed as operations — cloud_006 §3.2.
165
+ *
166
+ * Every method below is a **decision plus its write**, never a read the caller
167
+ * follows with a mutation. That is the whole point, and it is the difference
168
+ * between an interface a shared store can implement and one it cannot.
169
+ *
170
+ * `claim` is the specimen. It used to live in `DaemonPlane` as
171
+ * `jobs()` → filter → mutate, which is atomic for exactly one reason: Node is
172
+ * single-threaded and these Maps are local, so nothing runs between the read
173
+ * and the write. Neither survives a store on a network, and
174
+ * `packages/relay/test/two-replicas.test.ts` holds the resulting race as a
175
+ * failing assertion.
176
+ *
177
+ * So the rule for anything added here: **if a caller has to read, decide, and
178
+ * write back, the operation is in the wrong place.** Move the decision in.
179
+ *
180
+ * ## Why the projection does not come with it
181
+ *
182
+ * `claim` takes `owners: string[]` rather than a projection or a predicate.
183
+ * A closure cannot travel to Valkey, and the projection replicates for free
184
+ * from the control plane — so the caller collapses it with
185
+ * `Projection.ownersRunnableBy` and hands over data the store can match on.
186
+ * That keeps the store ignorant of consent, which is also what keeps it
187
+ * replaceable.
188
+ */
189
+ interface ClaimInput {
190
+ readonly runnerId: string;
191
+ readonly owner: string;
192
+ readonly device: PublicIdentity;
193
+ /**
194
+ * The (site, owner) pairs this device may run work for — cloud_009 §3.
195
+ *
196
+ * **One set of pairs, not a set of sites and a set of owners.** Consent
197
+ * binds a user to a site, so the two cannot travel separately: a device
198
+ * whose owner consented to site A, serving a roster member who consented
199
+ * to site B, would have every element of both sets and no consented route
200
+ * between them. Two sets multiply; consent does not.
201
+ *
202
+ * Built by {@link Projection.routesFor} and matched with {@link routeKey},
203
+ * so the relay and a store in another repository agree on the encoding by
204
+ * calling the same function rather than by both spelling it out.
205
+ *
206
+ * This is the collapse that lets a claim stay one operation: a predicate
207
+ * cannot travel to a store over a network, and a set can.
208
+ */
209
+ readonly routes: ReadonlySet<string>;
210
+ /**
211
+ * Kinds this device can run — and that is now the whole of the match.
212
+ *
213
+ * A relay routes by kind and by consent; **which service answers is not its
214
+ * question**. A job names a purpose, a person's mapping names a service,
215
+ * and a control plane joins them at claim. A relay that filtered on the
216
+ * service would need the mapping, which is the one thing it is not supposed
217
+ * to hold.
218
+ *
219
+ * This carried a companion, `serves`, holding the (kind, service) pairs a
220
+ * device advertised, so a job naming a service reached only a device with
221
+ * it. Sites stopped naming services (Amendment L) and the field became wire
222
+ * nothing read — so it is gone rather than left for a reader to infer
223
+ * meaning from.
224
+ *
225
+ * The cost is stated plainly: a job may be offered to a device whose owner
226
+ * admits the person but whose machine their mapping did not name. The
227
+ * control plane declines it as not-here and it goes back with a
228
+ * {@link RETRY_AFTER_MS} wait, which is what that mechanism is for.
229
+ */
230
+ readonly kinds: ReadonlySet<string>;
231
+ readonly max: number;
232
+ readonly leaseMs: number;
233
+ }
234
+ /**
235
+ * How a (site, owner) route is written, so two implementations agree.
236
+ *
237
+ * The same `\u0000` the job key uses, for the same reason: it cannot appear
238
+ * in a site id or an owner id, so this is a key rather than a parser.
239
+ */
240
+ declare const routeKey: (siteId: string, owner: string) => string;
241
+ /**
242
+ * Where the store's sense of time comes from — cloud_006 §3.4.
243
+ *
244
+ * **The store owns its clock; callers do not pass one.** Every deadline the
245
+ * relay decides — a lease's expiry, the `awaiting-payload` window, what a
246
+ * sweep considers due — is now stamped by one source, and it is the same
247
+ * source that will later stamp them for every replica.
248
+ *
249
+ * It used to be a parameter. `claim` took `now`, `sweep` took `now`, and each
250
+ * plane called its own `now()` before calling in — which is fine in one
251
+ * process and is the recurring bug the moment there are two. A lease granted
252
+ * by a pod whose clock runs fast is short; the same lease swept by a pod whose
253
+ * clock runs slow outlives it. Nobody is wrong and the lease has no length.
254
+ *
255
+ * A Valkey-backed store returns `TIME` here, so the deadline and the sweep
256
+ * that enforces it are read from the same server. The injected clock stays for
257
+ * tests, which is what lets them move time instead of sleeping.
258
+ *
259
+ * **What deliberately does not use this**: request-signature freshness. That
260
+ * is checked against the *local* clock on purpose — it is a question about the
261
+ * caller's clock versus this process's, `MAX_CLOCK_SKEW_MS` already tolerates
262
+ * two minutes of disagreement, and a network round trip to timestamp every
263
+ * inbound request would be a cost with no property behind it.
264
+ */
265
+ interface RelayStateOptions {
266
+ readonly now?: () => number | Promise<number>;
267
+ }
268
+ /** Why a lease-scoped operation was refused, in the caller's vocabulary. */
269
+ type HolderRefusal = "not-found" | "not-holder" | "stale-lease" | "not-ready"
270
+ /** The job already ended — V1-6. A replay must not reopen it. */
271
+ | "terminal";
272
+ declare class RelayState implements RoutingStore {
273
+ #private;
274
+ constructor(options?: RelayStateOptions);
275
+ /** The one clock every deadline in this store is stamped from. */
276
+ now(): Promise<number>;
277
+ /**
278
+ * Take a stub for routing. The payload is not here and will not be.
279
+ *
280
+ * **Idempotent by job id, and that is a security property rather than a
281
+ * convenience.** Site-plane calls are authenticated by signature, and
282
+ * byollm_009 §4.2's argument for signing the request instead of a
283
+ * server-issued nonce rests entirely on every write being idempotent per the
284
+ * instance it names. This one was not: re-enqueueing a known id built a
285
+ * fresh `queued` job over the top of the old one, discarding a live claim,
286
+ * its lease and any payload the site had already sealed to a device. A
287
+ * replayed enqueue inside the two-minute freshness window was therefore a
288
+ * way to yank a job back from the machine running it — the `release` bug of
289
+ * §4.2, rediscovered on the other plane.
290
+ *
291
+ * So a known id returns what is already routing, unchanged. A site that
292
+ * restarts and republishes its queue is the normal case, and it must not
293
+ * disturb work in flight.
294
+ */
295
+ enqueue(input: {
296
+ id: string;
297
+ siteId: string;
298
+ stub: JobStub;
299
+ }): Promise<RoutedJob>;
300
+ job(siteId: string, jobId: string): Promise<RoutedJob | undefined>;
301
+ jobs(): Promise<RoutedJob[]>;
302
+ /** Jobs a site must seal for, right now. */
303
+ awaiting(siteId: string): Promise<RoutedJob[]>;
304
+ /** Sealed results waiting to go home. */
305
+ finished(siteId: string): Promise<RoutedJob[]>;
306
+ /**
307
+ * Claim work — one operation, because it has to be.
308
+ *
309
+ * Moved here wholesale from `DaemonPlane`, where it was a scan followed by
310
+ * per-job mutation. Nothing about the *decision* changed; what changed is
311
+ * that a store can now implement it, because the filter and the write are
312
+ * one call rather than a loop the caller drives.
313
+ *
314
+ * The order of the guards is worth preserving as-is when this becomes a Lua
315
+ * script: cheapest first, and `owners` last because it is the only one that
316
+ * needed the projection.
317
+ */
318
+ claim(input: ClaimInput): Promise<ClaimedStub[]>;
319
+ /**
320
+ * Hand over the sealed payload to the device that holds the lease.
321
+ *
322
+ * The read and the state transition are one operation for the same reason
323
+ * `claim` is: `running` must be set by whoever was told the envelope, or two
324
+ * replicas can both hand out the same work and both believe they were first.
325
+ */
326
+ takePayload(input: {
327
+ jobId: string;
328
+ runnerId: string;
329
+ leaseId: string;
330
+ }): Promise<{
331
+ envelope: SealedEnvelope;
332
+ } | {
333
+ refused: HolderRefusal;
334
+ }>;
335
+ /**
336
+ * Record a finished job.
337
+ *
338
+ * `RESULT_IDEMPOTENT` lives here rather than in the caller: a replayed
339
+ * result must be a no-op decided by the same operation that would have
340
+ * written it, or two replicas can both decide they were the first.
341
+ */
342
+ complete(input: {
343
+ jobId: string;
344
+ runnerId: string;
345
+ leaseId: string;
346
+ envelope: SealedEnvelope;
347
+ disposition: "ok" | "error" | "canceled";
348
+ }): Promise<{
349
+ accepted: boolean;
350
+ duplicate?: boolean;
351
+ state: RoutedState;
352
+ } | {
353
+ refused: HolderRefusal;
354
+ }>;
355
+ /** Give back leases this runner holds, naming each grant it means. */
356
+ releaseLeases(input: {
357
+ runnerId: string;
358
+ leases: readonly {
359
+ jobId: string;
360
+ leaseId: string;
361
+ }[];
362
+ reason?: ReleaseReason;
363
+ retryAfter?: number;
364
+ }): Promise<string[]>;
365
+ /**
366
+ * Take a site's sealed payload for a claimed job.
367
+ *
368
+ * Refuses anything not `awaiting-payload`, which is what makes the timeout
369
+ * mean something: a late seal must not land on a claim that has moved.
370
+ */
371
+ seal(input: {
372
+ jobId: string;
373
+ siteId: string;
374
+ envelope: SealedEnvelope;
375
+ }): Promise<{
376
+ state: RoutedState;
377
+ } | {
378
+ refused: "not-found" | "too-late";
379
+ was?: RoutedState;
380
+ }>;
381
+ /** {@link RoutingStore.cancel} — the site withdraws a job. */
382
+ cancel(input: {
383
+ jobId: string;
384
+ siteId: string;
385
+ }): Promise<boolean>;
386
+ /** {@link RoutingStore.cancelRequests} — cancelled jobs this runner holds. */
387
+ cancelRequests(runnerId: string): Promise<Grant[]>;
388
+ /** {@link RoutingStore.renewLeases} — extend what is still held, name what is not. */
389
+ renewLeases(input: {
390
+ runnerId: string;
391
+ leases: readonly {
392
+ jobId: string;
393
+ leaseId: string;
394
+ }[];
395
+ leaseMs: number;
396
+ }): Promise<{
397
+ renewed: {
398
+ jobId: string;
399
+ expiresAt: number;
400
+ }[];
401
+ lost: Grant[];
402
+ }>;
403
+ seen(presence: Omit<Presence, "lastSeenAt">): Promise<Presence>;
404
+ presence(runnerId: string): Promise<Presence | undefined>;
405
+ /**
406
+ * Lose a record, the way a real store does.
407
+ *
408
+ * A shared store drops presence for reasons this one never will — a TTL, a
409
+ * reschedule, a restart — and the interesting behaviour is what the relay
410
+ * does next. `ValkeyRoutingStore` has carried the same helper since
411
+ * finding 52; this is its memory twin, so the case can be written once
412
+ * against the implementation that is easy to reason about.
413
+ */
414
+ dropPresenceForTests(runnerId: string): Promise<void>;
415
+ everyone(): Promise<Presence[]>;
416
+ /**
417
+ * Fire whatever the clock says is due, and report it.
418
+ *
419
+ * Returns the jobs it requeued so a caller can log or surface them — a
420
+ * timeout that fires invisibly is indistinguishable from a job that was
421
+ * never claimed, and those want very different debugging.
422
+ */
423
+ sweep(): Promise<RoutedJob[]>;
424
+ }
425
+
426
+ /**
427
+ * What the relay needs from a place to keep routing state — cloud_006 §3.2.
428
+ *
429
+ * `RelayState` implements this in memory and is the reference; a Valkey-backed
430
+ * store implements the same thing across replicas. The interface exists so the
431
+ * relay depends on the *contract* rather than on either, and so the properties
432
+ * below are stated once rather than rediscovered per implementation.
433
+ *
434
+ * ## Every method is a decision plus its write
435
+ *
436
+ * Not one is a read the caller follows with a mutation. That is the whole
437
+ * design, and it is not a style preference: `claim` used to be
438
+ * `jobs()` → filter → mutate in the plane, which is atomic for exactly one
439
+ * reason — Node is single-threaded and the Maps are local. Neither survives a
440
+ * store on a network, and `packages/relay/test/two-replicas.test.ts` holds the
441
+ * resulting race as a failing assertion.
442
+ *
443
+ * **The rule for anything added here:** if a caller has to read, decide, and
444
+ * write back, the operation is in the wrong place. Move the decision in.
445
+ *
446
+ * ## What an implementation must guarantee
447
+ *
448
+ * 1. **`claim` is atomic.** Two callers claiming concurrently must not both
449
+ * receive the same job. `CLAIM_ATOMIC` is a protocol MUST.
450
+ * 2. **`enqueue` is idempotent by job id.** A known id returns what is already
451
+ * routing rather than rebuilding it — byollm_009 §4.2's replay argument
452
+ * rests on every write being idempotent per the instance it names.
453
+ * 3. **`complete` is idempotent.** A replayed result changes nothing, and the
454
+ * decision is made by the same operation that would have written it.
455
+ * 4. **Lease-scoped operations name the grant.** `takePayload`, `complete` and
456
+ * `releaseLeases` check the lease *id*, not just the runner — a runner
457
+ * survives a claim-release-reclaim cycle and a grant does not.
458
+ * 5. **`now()` is the only clock.** Every deadline the store stamps and every
459
+ * deadline it enforces come from here (§3.4). An implementation backed by a
460
+ * server returns that server's time, so two replicas cannot disagree about
461
+ * how long a lease is.
462
+ *
463
+ * ## What it must not do
464
+ *
465
+ * Hold a key, or learn about consent. `claim` takes `owners` as data because a
466
+ * predicate cannot travel to Valkey — and the effect is that the store cannot
467
+ * express an opinion about who may route, only about what it was told. That is
468
+ * what keeps `RELAY_BLIND` a property of the shape rather than of the code.
469
+ */
470
+ /**
471
+ * One grant, named by both halves — V1-3.
472
+ *
473
+ * A job id is chosen per site, so two sites can pick the same one; the lease
474
+ * id is the unique thing an upstream and a daemon can both point at. Anywhere
475
+ * this store says "this piece of work, held by you", it says it with both.
476
+ */
477
+ interface Grant {
478
+ readonly jobId: string;
479
+ readonly leaseId: string;
480
+ }
481
+ interface RoutingStore {
482
+ /** The one clock every deadline in this store is stamped from. */
483
+ now(): Promise<number>;
484
+ /**
485
+ * Take a stub for routing. Idempotent by id, **within a site**.
486
+ *
487
+ * A republished id belonging to the same site returns what is already
488
+ * routing, unchanged: a site that restarts and republishes its queue must
489
+ * not disturb work in flight (byollm_009 §4.2's replay argument rests on
490
+ * it).
491
+ *
492
+ * A republished id belonging to a **different** site is a different job —
493
+ * cloud_009 §3. Keys are `(site, id)`, so two sites choosing one id collide
494
+ * nowhere and there is nothing to refuse.
495
+ *
496
+ * cloud_008 finding 58 closed this by refusing the collision, which was as
497
+ * far as a single-site relay could go and left a cross-tenant existence
498
+ * oracle: a site knows its own stub is well-formed, so any answer it can
499
+ * tell apart from success confirms that somebody else holds that id. The
500
+ * refusal is gone with the collision.
501
+ */
502
+ enqueue(input: {
503
+ id: string;
504
+ siteId: string;
505
+ stub: JobStub;
506
+ }): Promise<RoutedJob>;
507
+ /**
508
+ * One job, named by the site that published it — cloud_009 §3.
509
+ *
510
+ * The site id is a parameter rather than a scan, and the difference is
511
+ * finding eleven wearing a different hat: a reader that finds a job by id
512
+ * alone is a reader that can see every tenant's state through one door,
513
+ * which is exactly the anonymous read the debug page had and lost. The
514
+ * debug page is per-site or it is nothing.
515
+ */
516
+ job(siteId: string, jobId: string): Promise<RoutedJob | undefined>;
517
+ jobs(): Promise<RoutedJob[]>;
518
+ /** Jobs a site must seal for, right now. */
519
+ awaiting(siteId: string): Promise<RoutedJob[]>;
520
+ /** Sealed results waiting to go home. */
521
+ finished(siteId: string): Promise<RoutedJob[]>;
522
+ /** Grant work to a device — one operation, because it has to be. */
523
+ claim(input: ClaimInput): Promise<ClaimedStub[]>;
524
+ /** Hand the sealed payload to the device that holds the lease. */
525
+ takePayload(input: {
526
+ jobId: string;
527
+ runnerId: string;
528
+ leaseId: string;
529
+ }): Promise<{
530
+ envelope: SealedEnvelope;
531
+ } | {
532
+ refused: HolderRefusal;
533
+ }>;
534
+ /** Record a finished job. Idempotent. */
535
+ complete(input: {
536
+ jobId: string;
537
+ runnerId: string;
538
+ /**
539
+ * The grant the result was produced under — cloud_008 §1.4a.
540
+ *
541
+ * `takePayload` and `releaseLeases` have always checked this and said why;
542
+ * the operation that *writes* checked only the runner, which survives a
543
+ * claim-release-reclaim cycle.
544
+ */
545
+ leaseId: string;
546
+ envelope: SealedEnvelope;
547
+ disposition: "ok" | "error" | "canceled";
548
+ }): Promise<{
549
+ accepted: boolean;
550
+ duplicate?: boolean;
551
+ state: RoutedState;
552
+ } | {
553
+ refused: HolderRefusal;
554
+ }>;
555
+ /** Give back the grants this runner names. */
556
+ releaseLeases(input: {
557
+ runnerId: string;
558
+ leases: readonly {
559
+ jobId: string;
560
+ leaseId: string;
561
+ }[];
562
+ /**
563
+ * Why — cloud_008 §2.1. `refused` MUST be remembered and that job MUST
564
+ * NOT be offered to that runner again (`REFUSAL_NOT_REOFFERED`); every
565
+ * other reason means "not now" and must leave the job claimable by the
566
+ * same device, or a restart would strand its own work.
567
+ */
568
+ reason?: ReleaseReason;
569
+ /**
570
+ * Epoch ms before which this job MUST NOT be offered to this runner
571
+ * again — the rate a transient refusal needs.
572
+ *
573
+ * "Not now" and "not ever" were the only two things a release could say,
574
+ * and a control plane declining a job for something the world can change
575
+ * means neither. Left immediate, the device re-claims at once and is
576
+ * declined again: a spin, and in a deployment one database read per turn
577
+ * of it.
578
+ *
579
+ * A store MUST NOT offer the job to that runner before this moment, and
580
+ * MUST go on offering it to every other device immediately — the whole
581
+ * reason it is not a refusal is that another machine may be the right
582
+ * one.
583
+ */
584
+ retryAfter?: number;
585
+ }): Promise<string[]>;
586
+ /** Take a site's sealed payload for a claimed job. */
587
+ seal(input: {
588
+ jobId: string;
589
+ siteId: string;
590
+ envelope: SealedEnvelope;
591
+ }): Promise<{
592
+ state: RoutedState;
593
+ } | {
594
+ refused: "not-found" | "too-late";
595
+ was?: RoutedState;
596
+ }>;
597
+ /**
598
+ * Extend the leases this runner still holds, and name the ones it does not.
599
+ *
600
+ * One call because it is one question — cloud_008 §0.6. This was
601
+ * `lostLeases`, which answered only the second half, and the relay's
602
+ * heartbeat answered the first half with the literal `leases: []`. A daemon
603
+ * was therefore told, every few seconds, that none of its work had been
604
+ * renewed; the sweep requeued at `leaseMs` regardless of how alive the
605
+ * device was, and any job that took longer than a lease was handed to
606
+ * somebody else while the first device was still running it. The direct
607
+ * plane has always renewed here (`handlers.ts` §3), so this was also the two
608
+ * upstreams disagreeing about a rule the daemon cannot see.
609
+ *
610
+ * Renewal and loss come from one read of one state: asked separately they
611
+ * are two answers to "who holds this now", and under two replicas they can
612
+ * differ.
613
+ */
614
+ renewLeases(input: {
615
+ runnerId: string;
616
+ leases: readonly {
617
+ jobId: string;
618
+ leaseId: string;
619
+ }[];
620
+ leaseMs: number;
621
+ }): Promise<{
622
+ renewed: readonly {
623
+ jobId: string;
624
+ expiresAt: number;
625
+ }[];
626
+ lost: readonly Grant[];
627
+ }>;
628
+ /**
629
+ * The site withdraws a job — cloud_008 §2.2.
630
+ *
631
+ * Returns false when the job is not this site's, which is the same scoping
632
+ * every other site-plane operation carries: a site must not cancel
633
+ * somebody else's work by guessing an id.
634
+ */
635
+ cancel(input: {
636
+ jobId: string;
637
+ siteId: string;
638
+ }): Promise<boolean>;
639
+ /**
640
+ * Cancelled jobs this runner is holding, for the heartbeat to report.
641
+ *
642
+ * The relay answered `cancel: []` unconditionally, so a site could not stop
643
+ * a job it had already withdrawn — a device went on running work whose
644
+ * result nobody would accept, on somebody's own machine, at their expense.
645
+ */
646
+ cancelRequests(runnerId: string): Promise<Grant[]>;
647
+ /** Record a device as present. The store stamps when. */
648
+ seen(presence: Omit<Presence, "lastSeenAt">): Promise<Presence>;
649
+ presence(runnerId: string): Promise<Presence | undefined>;
650
+ everyone(): Promise<Presence[]>;
651
+ /** Fire whatever the clock says is due, and report it. */
652
+ sweep(): Promise<RoutedJob[]>;
653
+ }
654
+
655
+ export { AWAITING_PAYLOAD_MS as A, type ClaimInput as C, type Grant as G, type HolderRefusal as H, type Presence as P, type RoutingStore as R, RelayState as a, type ReleaseReason as b, type RoutedJob as c, type RoutedState as d, routeKey as r };