@byollm/relay 0.1.0-alpha.7 → 0.1.0-alpha.71
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +166 -4
- package/dist/chunk-OB6LPEEE.js +405 -0
- package/dist/chunk-OB6LPEEE.js.map +1 -0
- package/dist/index.d.ts +507 -153
- package/dist/index.js +1069 -272
- package/dist/index.js.map +1 -1
- package/dist/store-DPCLO12l.d.ts +655 -0
- package/dist/store-contract.d.ts +67 -0
- package/dist/store-contract.js +625 -0
- package/dist/store-contract.js.map +1 -0
- package/package.json +19 -4
|
@@ -0,0 +1,655 @@
|
|
|
1
|
+
import { PublicIdentity, CapabilityMatrix, WithheldKind, JobStub, SealedEnvelope, ClaimedStub } from '@byollm/protocol';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* The relay's routing state — byollm_009 §7, reachable at last.
|
|
5
|
+
*
|
|
6
|
+
* §7 described a state machine the direct plane could not produce. There, the
|
|
7
|
+
* site and the upstream are the same party: it seals when it likes, and a job
|
|
8
|
+
* is never claimed-but-unsealed. Here they are different parties, and the gap
|
|
9
|
+
* between them is a state:
|
|
10
|
+
*
|
|
11
|
+
* ```
|
|
12
|
+
* queued ──claim──▶ awaiting-payload ──sealed──▶ ready ──fetch──▶ running
|
|
13
|
+
* ▲ │ │
|
|
14
|
+
* └────────────────────┘ ▼
|
|
15
|
+
* site never seals, or seals too late ok | error | canceled
|
|
16
|
+
* ```
|
|
17
|
+
*
|
|
18
|
+
* The relay cannot seal, so it cannot shortcut this. A payload is encrypted
|
|
19
|
+
* to *the device that claimed it*, and nobody knows which device that is until
|
|
20
|
+
* the claim happens — which is precisely why claim-then-fetch makes a blind
|
|
21
|
+
* relay possible at all. The window is the price.
|
|
22
|
+
*
|
|
23
|
+
* ## What the relay holds, and what it cannot
|
|
24
|
+
*
|
|
25
|
+
* Stubs (metadata the site chose to publish), sealed envelopes it cannot open,
|
|
26
|
+
* and public keys. There is no field on any type in this file that could hold
|
|
27
|
+
* a private key or a plaintext, which is `RELAY_BLIND` expressed as a data
|
|
28
|
+
* model rather than as a policy.
|
|
29
|
+
*/
|
|
30
|
+
/** Where a routed job is. */
|
|
31
|
+
type RoutedState = "queued" | "awaiting-payload" | "ready" | "running" | "done";
|
|
32
|
+
/**
|
|
33
|
+
* How long a site has to seal after one of its jobs is claimed.
|
|
34
|
+
*
|
|
35
|
+
* **Distinct from the lease, and distinct from the job's TTL** — byollm_009
|
|
36
|
+
* §7.1. Three clocks, three different questions:
|
|
37
|
+
*
|
|
38
|
+
* - the **TTL** asks how long the work is worth doing at all;
|
|
39
|
+
* - the **lease** asks how long this device gets to run it;
|
|
40
|
+
* - this asks how long we wait for a site that has gone away.
|
|
41
|
+
*
|
|
42
|
+
* Collapsing any pair of them looks harmless until a site restarts during a
|
|
43
|
+
* deploy: with only a lease, the device sits politely holding a job whose
|
|
44
|
+
* payload will never arrive, and the lease's whole minute is spent waiting on
|
|
45
|
+
* a party that is not coming back. Short, because a site that is up answers in
|
|
46
|
+
* milliseconds and a site that is down will not answer sooner for waiting.
|
|
47
|
+
*/
|
|
48
|
+
declare const AWAITING_PAYLOAD_MS = 10000;
|
|
49
|
+
/** A job the relay is routing. Metadata and ciphertext, nothing else. */
|
|
50
|
+
/** Why a daemon gave a job back. Only `refused` means "not me, ever". */
|
|
51
|
+
type ReleaseReason = "shutdown" | "pause" | "revoked" | "backend-down" | "refused";
|
|
52
|
+
interface RoutedJob {
|
|
53
|
+
readonly id: string;
|
|
54
|
+
/** Which site enqueued it — the party that will be asked to seal. */
|
|
55
|
+
readonly siteId: string;
|
|
56
|
+
/**
|
|
57
|
+
* Everything the relay knows about the work, which is everything the site
|
|
58
|
+
* chose to publish and not one field more (byollm_009 §6).
|
|
59
|
+
*/
|
|
60
|
+
readonly stub: JobStub;
|
|
61
|
+
state: RoutedState;
|
|
62
|
+
/** Set from the claim; the site seals to these keys. */
|
|
63
|
+
claimedBy?: {
|
|
64
|
+
readonly runnerId: string;
|
|
65
|
+
readonly owner: string;
|
|
66
|
+
readonly device: PublicIdentity;
|
|
67
|
+
readonly leaseId: string;
|
|
68
|
+
readonly leaseExpiresAt: number;
|
|
69
|
+
};
|
|
70
|
+
/** When {@link AWAITING_PAYLOAD_MS} runs out for this claim. */
|
|
71
|
+
awaitingUntil?: number;
|
|
72
|
+
/**
|
|
73
|
+
* Runners that released this job with reason `refused` — cloud_008 §2.1.
|
|
74
|
+
*
|
|
75
|
+
* `REFUSAL_NOT_REOFFERED`, which the relay did not implement: it dropped
|
|
76
|
+
* `ReleaseRequest.reason` on the floor. The field's own docstring says why
|
|
77
|
+
* that is not cosmetic — an upstream cannot evaluate a daemon's *local*
|
|
78
|
+
* `named` allowlist, so it may legitimately offer work the daemon then
|
|
79
|
+
* declines, and without a record the two spin between claim and release
|
|
80
|
+
* forever. The direct plane has always kept this list.
|
|
81
|
+
*/
|
|
82
|
+
refusedBy: string[];
|
|
83
|
+
/**
|
|
84
|
+
* Runners that may not be offered this job again *yet*, and from when.
|
|
85
|
+
*
|
|
86
|
+
* The middle ground {@link RoutingStore.releaseLeases} had no way to say.
|
|
87
|
+
* `refusedBy` is forever and a bare release is immediate; a control plane
|
|
88
|
+
* declining a job for a reason the world can change — an unfilled mapping
|
|
89
|
+
* slot, a resolution that named another machine, a store that was briefly
|
|
90
|
+
* unreachable — means neither. It means "ask again later", and later needs
|
|
91
|
+
* a number.
|
|
92
|
+
*
|
|
93
|
+
* Keyed by runner because it is a fact about a pairing, not about the job:
|
|
94
|
+
* the same job goes to another device immediately, which is the whole
|
|
95
|
+
* point of not marking it refused.
|
|
96
|
+
*/
|
|
97
|
+
retryAfter?: Record<string, number>;
|
|
98
|
+
/**
|
|
99
|
+
* The site withdrew this job — cloud_008 §2.2.
|
|
100
|
+
*
|
|
101
|
+
* A flag rather than a state, because a cancelled job that a device is
|
|
102
|
+
* *running* is not finished: the daemon has to be told, abort its backend
|
|
103
|
+
* call and report `canceled`, and the ordinary `complete` path then closes
|
|
104
|
+
* it. Making it a state would strand the in-flight case between two
|
|
105
|
+
* machines' ideas of what happened.
|
|
106
|
+
*/
|
|
107
|
+
cancelled?: boolean;
|
|
108
|
+
/** Sealed to the claiming device by the site. Opaque here. */
|
|
109
|
+
payload?: SealedEnvelope;
|
|
110
|
+
/** Sealed to the site by the device. Opaque here. */
|
|
111
|
+
result?: SealedEnvelope;
|
|
112
|
+
/**
|
|
113
|
+
* The result's clear-text discriminator — byollm_009 §6.1.
|
|
114
|
+
*
|
|
115
|
+
* The one outcome fact the relay is given, and the reason it is given:
|
|
116
|
+
* without it the relay cannot stop dispatching a finished job. A routing
|
|
117
|
+
* hint and never a fact — the *site* verifies it against the sealed
|
|
118
|
+
* outcome, because only the site can open the envelope. The relay acts on
|
|
119
|
+
* it and is entitled to be wrong; a lying daemon costs it a dispatch
|
|
120
|
+
* decision, not a security property.
|
|
121
|
+
*/
|
|
122
|
+
disposition?: "ok" | "error" | "canceled";
|
|
123
|
+
}
|
|
124
|
+
/** A device the relay has seen recently. */
|
|
125
|
+
interface Presence {
|
|
126
|
+
readonly runnerId: string;
|
|
127
|
+
readonly owner: string;
|
|
128
|
+
readonly device: PublicIdentity;
|
|
129
|
+
lastSeenAt: number;
|
|
130
|
+
/**
|
|
131
|
+
* What this machine last said it can run — cloud_009, 2026-08-24.
|
|
132
|
+
*
|
|
133
|
+
* **Capabilities are presence data.** They arrive on the same heartbeat as
|
|
134
|
+
* everything else here, they go stale at the same moment and for the same
|
|
135
|
+
* reason, and a machine that stops heartbeating has not stopped being able
|
|
136
|
+
* to run Llama — it has stopped being somewhere we can ask. Keeping them
|
|
137
|
+
* anywhere else would put capability truth outside the interface that owns
|
|
138
|
+
* presence, and every next consumer (a member's usable set, a dashboard's
|
|
139
|
+
* pulse, a degraded-state banner) would re-derive it from a different
|
|
140
|
+
* place. That is the two-owners shape this codebase keeps paying for.
|
|
141
|
+
*
|
|
142
|
+
* Empty is a real answer, not a missing one: it is a paired machine with no
|
|
143
|
+
* healthy backend, which is a legal state the whole connect-first ruling
|
|
144
|
+
* exists to make visible rather than refuse.
|
|
145
|
+
*/
|
|
146
|
+
capabilities: CapabilityMatrix;
|
|
147
|
+
/**
|
|
148
|
+
* Kinds this machine is deliberately *not* advertising — byollm_016.
|
|
149
|
+
*
|
|
150
|
+
* Presence data for the same reason capabilities are, and stored beside
|
|
151
|
+
* them rather than derived: two services answering one kind with no
|
|
152
|
+
* `defaults` entry is a state only the daemon can see, and the hub cannot
|
|
153
|
+
* reconstruct it from the matrix — an absent kind and a withheld kind look
|
|
154
|
+
* identical there. Without this field the owner's page can say "nothing
|
|
155
|
+
* serves llm.generate", which is true and useless, instead of "two services
|
|
156
|
+
* answer it and you have not chosen", which is the sentence that ends with
|
|
157
|
+
* the owner doing something.
|
|
158
|
+
*
|
|
159
|
+
* Empty is the normal answer, and means every kind resolved.
|
|
160
|
+
*/
|
|
161
|
+
withheld: readonly WithheldKind[];
|
|
162
|
+
}
|
|
163
|
+
/**
|
|
164
|
+
* What a routing store must do, expressed as operations — cloud_006 §3.2.
|
|
165
|
+
*
|
|
166
|
+
* Every method below is a **decision plus its write**, never a read the caller
|
|
167
|
+
* follows with a mutation. That is the whole point, and it is the difference
|
|
168
|
+
* between an interface a shared store can implement and one it cannot.
|
|
169
|
+
*
|
|
170
|
+
* `claim` is the specimen. It used to live in `DaemonPlane` as
|
|
171
|
+
* `jobs()` → filter → mutate, which is atomic for exactly one reason: Node is
|
|
172
|
+
* single-threaded and these Maps are local, so nothing runs between the read
|
|
173
|
+
* and the write. Neither survives a store on a network, and
|
|
174
|
+
* `packages/relay/test/two-replicas.test.ts` holds the resulting race as a
|
|
175
|
+
* failing assertion.
|
|
176
|
+
*
|
|
177
|
+
* So the rule for anything added here: **if a caller has to read, decide, and
|
|
178
|
+
* write back, the operation is in the wrong place.** Move the decision in.
|
|
179
|
+
*
|
|
180
|
+
* ## Why the projection does not come with it
|
|
181
|
+
*
|
|
182
|
+
* `claim` takes `owners: string[]` rather than a projection or a predicate.
|
|
183
|
+
* A closure cannot travel to Valkey, and the projection replicates for free
|
|
184
|
+
* from the control plane — so the caller collapses it with
|
|
185
|
+
* `Projection.ownersRunnableBy` and hands over data the store can match on.
|
|
186
|
+
* That keeps the store ignorant of consent, which is also what keeps it
|
|
187
|
+
* replaceable.
|
|
188
|
+
*/
|
|
189
|
+
interface ClaimInput {
|
|
190
|
+
readonly runnerId: string;
|
|
191
|
+
readonly owner: string;
|
|
192
|
+
readonly device: PublicIdentity;
|
|
193
|
+
/**
|
|
194
|
+
* The (site, owner) pairs this device may run work for — cloud_009 §3.
|
|
195
|
+
*
|
|
196
|
+
* **One set of pairs, not a set of sites and a set of owners.** Consent
|
|
197
|
+
* binds a user to a site, so the two cannot travel separately: a device
|
|
198
|
+
* whose owner consented to site A, serving a roster member who consented
|
|
199
|
+
* to site B, would have every element of both sets and no consented route
|
|
200
|
+
* between them. Two sets multiply; consent does not.
|
|
201
|
+
*
|
|
202
|
+
* Built by {@link Projection.routesFor} and matched with {@link routeKey},
|
|
203
|
+
* so the relay and a store in another repository agree on the encoding by
|
|
204
|
+
* calling the same function rather than by both spelling it out.
|
|
205
|
+
*
|
|
206
|
+
* This is the collapse that lets a claim stay one operation: a predicate
|
|
207
|
+
* cannot travel to a store over a network, and a set can.
|
|
208
|
+
*/
|
|
209
|
+
readonly routes: ReadonlySet<string>;
|
|
210
|
+
/**
|
|
211
|
+
* Kinds this device can run — and that is now the whole of the match.
|
|
212
|
+
*
|
|
213
|
+
* A relay routes by kind and by consent; **which service answers is not its
|
|
214
|
+
* question**. A job names a purpose, a person's mapping names a service,
|
|
215
|
+
* and a control plane joins them at claim. A relay that filtered on the
|
|
216
|
+
* service would need the mapping, which is the one thing it is not supposed
|
|
217
|
+
* to hold.
|
|
218
|
+
*
|
|
219
|
+
* This carried a companion, `serves`, holding the (kind, service) pairs a
|
|
220
|
+
* device advertised, so a job naming a service reached only a device with
|
|
221
|
+
* it. Sites stopped naming services (Amendment L) and the field became wire
|
|
222
|
+
* nothing read — so it is gone rather than left for a reader to infer
|
|
223
|
+
* meaning from.
|
|
224
|
+
*
|
|
225
|
+
* The cost is stated plainly: a job may be offered to a device whose owner
|
|
226
|
+
* admits the person but whose machine their mapping did not name. The
|
|
227
|
+
* control plane declines it as not-here and it goes back with a
|
|
228
|
+
* {@link RETRY_AFTER_MS} wait, which is what that mechanism is for.
|
|
229
|
+
*/
|
|
230
|
+
readonly kinds: ReadonlySet<string>;
|
|
231
|
+
readonly max: number;
|
|
232
|
+
readonly leaseMs: number;
|
|
233
|
+
}
|
|
234
|
+
/**
|
|
235
|
+
* How a (site, owner) route is written, so two implementations agree.
|
|
236
|
+
*
|
|
237
|
+
* The same `\u0000` the job key uses, for the same reason: it cannot appear
|
|
238
|
+
* in a site id or an owner id, so this is a key rather than a parser.
|
|
239
|
+
*/
|
|
240
|
+
declare const routeKey: (siteId: string, owner: string) => string;
|
|
241
|
+
/**
|
|
242
|
+
* Where the store's sense of time comes from — cloud_006 §3.4.
|
|
243
|
+
*
|
|
244
|
+
* **The store owns its clock; callers do not pass one.** Every deadline the
|
|
245
|
+
* relay decides — a lease's expiry, the `awaiting-payload` window, what a
|
|
246
|
+
* sweep considers due — is now stamped by one source, and it is the same
|
|
247
|
+
* source that will later stamp them for every replica.
|
|
248
|
+
*
|
|
249
|
+
* It used to be a parameter. `claim` took `now`, `sweep` took `now`, and each
|
|
250
|
+
* plane called its own `now()` before calling in — which is fine in one
|
|
251
|
+
* process and is the recurring bug the moment there are two. A lease granted
|
|
252
|
+
* by a pod whose clock runs fast is short; the same lease swept by a pod whose
|
|
253
|
+
* clock runs slow outlives it. Nobody is wrong and the lease has no length.
|
|
254
|
+
*
|
|
255
|
+
* A Valkey-backed store returns `TIME` here, so the deadline and the sweep
|
|
256
|
+
* that enforces it are read from the same server. The injected clock stays for
|
|
257
|
+
* tests, which is what lets them move time instead of sleeping.
|
|
258
|
+
*
|
|
259
|
+
* **What deliberately does not use this**: request-signature freshness. That
|
|
260
|
+
* is checked against the *local* clock on purpose — it is a question about the
|
|
261
|
+
* caller's clock versus this process's, `MAX_CLOCK_SKEW_MS` already tolerates
|
|
262
|
+
* two minutes of disagreement, and a network round trip to timestamp every
|
|
263
|
+
* inbound request would be a cost with no property behind it.
|
|
264
|
+
*/
|
|
265
|
+
interface RelayStateOptions {
|
|
266
|
+
readonly now?: () => number | Promise<number>;
|
|
267
|
+
}
|
|
268
|
+
/** Why a lease-scoped operation was refused, in the caller's vocabulary. */
|
|
269
|
+
type HolderRefusal = "not-found" | "not-holder" | "stale-lease" | "not-ready"
|
|
270
|
+
/** The job already ended — V1-6. A replay must not reopen it. */
|
|
271
|
+
| "terminal";
|
|
272
|
+
declare class RelayState implements RoutingStore {
|
|
273
|
+
#private;
|
|
274
|
+
constructor(options?: RelayStateOptions);
|
|
275
|
+
/** The one clock every deadline in this store is stamped from. */
|
|
276
|
+
now(): Promise<number>;
|
|
277
|
+
/**
|
|
278
|
+
* Take a stub for routing. The payload is not here and will not be.
|
|
279
|
+
*
|
|
280
|
+
* **Idempotent by job id, and that is a security property rather than a
|
|
281
|
+
* convenience.** Site-plane calls are authenticated by signature, and
|
|
282
|
+
* byollm_009 §4.2's argument for signing the request instead of a
|
|
283
|
+
* server-issued nonce rests entirely on every write being idempotent per the
|
|
284
|
+
* instance it names. This one was not: re-enqueueing a known id built a
|
|
285
|
+
* fresh `queued` job over the top of the old one, discarding a live claim,
|
|
286
|
+
* its lease and any payload the site had already sealed to a device. A
|
|
287
|
+
* replayed enqueue inside the two-minute freshness window was therefore a
|
|
288
|
+
* way to yank a job back from the machine running it — the `release` bug of
|
|
289
|
+
* §4.2, rediscovered on the other plane.
|
|
290
|
+
*
|
|
291
|
+
* So a known id returns what is already routing, unchanged. A site that
|
|
292
|
+
* restarts and republishes its queue is the normal case, and it must not
|
|
293
|
+
* disturb work in flight.
|
|
294
|
+
*/
|
|
295
|
+
enqueue(input: {
|
|
296
|
+
id: string;
|
|
297
|
+
siteId: string;
|
|
298
|
+
stub: JobStub;
|
|
299
|
+
}): Promise<RoutedJob>;
|
|
300
|
+
job(siteId: string, jobId: string): Promise<RoutedJob | undefined>;
|
|
301
|
+
jobs(): Promise<RoutedJob[]>;
|
|
302
|
+
/** Jobs a site must seal for, right now. */
|
|
303
|
+
awaiting(siteId: string): Promise<RoutedJob[]>;
|
|
304
|
+
/** Sealed results waiting to go home. */
|
|
305
|
+
finished(siteId: string): Promise<RoutedJob[]>;
|
|
306
|
+
/**
|
|
307
|
+
* Claim work — one operation, because it has to be.
|
|
308
|
+
*
|
|
309
|
+
* Moved here wholesale from `DaemonPlane`, where it was a scan followed by
|
|
310
|
+
* per-job mutation. Nothing about the *decision* changed; what changed is
|
|
311
|
+
* that a store can now implement it, because the filter and the write are
|
|
312
|
+
* one call rather than a loop the caller drives.
|
|
313
|
+
*
|
|
314
|
+
* The order of the guards is worth preserving as-is when this becomes a Lua
|
|
315
|
+
* script: cheapest first, and `owners` last because it is the only one that
|
|
316
|
+
* needed the projection.
|
|
317
|
+
*/
|
|
318
|
+
claim(input: ClaimInput): Promise<ClaimedStub[]>;
|
|
319
|
+
/**
|
|
320
|
+
* Hand over the sealed payload to the device that holds the lease.
|
|
321
|
+
*
|
|
322
|
+
* The read and the state transition are one operation for the same reason
|
|
323
|
+
* `claim` is: `running` must be set by whoever was told the envelope, or two
|
|
324
|
+
* replicas can both hand out the same work and both believe they were first.
|
|
325
|
+
*/
|
|
326
|
+
takePayload(input: {
|
|
327
|
+
jobId: string;
|
|
328
|
+
runnerId: string;
|
|
329
|
+
leaseId: string;
|
|
330
|
+
}): Promise<{
|
|
331
|
+
envelope: SealedEnvelope;
|
|
332
|
+
} | {
|
|
333
|
+
refused: HolderRefusal;
|
|
334
|
+
}>;
|
|
335
|
+
/**
|
|
336
|
+
* Record a finished job.
|
|
337
|
+
*
|
|
338
|
+
* `RESULT_IDEMPOTENT` lives here rather than in the caller: a replayed
|
|
339
|
+
* result must be a no-op decided by the same operation that would have
|
|
340
|
+
* written it, or two replicas can both decide they were the first.
|
|
341
|
+
*/
|
|
342
|
+
complete(input: {
|
|
343
|
+
jobId: string;
|
|
344
|
+
runnerId: string;
|
|
345
|
+
leaseId: string;
|
|
346
|
+
envelope: SealedEnvelope;
|
|
347
|
+
disposition: "ok" | "error" | "canceled";
|
|
348
|
+
}): Promise<{
|
|
349
|
+
accepted: boolean;
|
|
350
|
+
duplicate?: boolean;
|
|
351
|
+
state: RoutedState;
|
|
352
|
+
} | {
|
|
353
|
+
refused: HolderRefusal;
|
|
354
|
+
}>;
|
|
355
|
+
/** Give back leases this runner holds, naming each grant it means. */
|
|
356
|
+
releaseLeases(input: {
|
|
357
|
+
runnerId: string;
|
|
358
|
+
leases: readonly {
|
|
359
|
+
jobId: string;
|
|
360
|
+
leaseId: string;
|
|
361
|
+
}[];
|
|
362
|
+
reason?: ReleaseReason;
|
|
363
|
+
retryAfter?: number;
|
|
364
|
+
}): Promise<string[]>;
|
|
365
|
+
/**
|
|
366
|
+
* Take a site's sealed payload for a claimed job.
|
|
367
|
+
*
|
|
368
|
+
* Refuses anything not `awaiting-payload`, which is what makes the timeout
|
|
369
|
+
* mean something: a late seal must not land on a claim that has moved.
|
|
370
|
+
*/
|
|
371
|
+
seal(input: {
|
|
372
|
+
jobId: string;
|
|
373
|
+
siteId: string;
|
|
374
|
+
envelope: SealedEnvelope;
|
|
375
|
+
}): Promise<{
|
|
376
|
+
state: RoutedState;
|
|
377
|
+
} | {
|
|
378
|
+
refused: "not-found" | "too-late";
|
|
379
|
+
was?: RoutedState;
|
|
380
|
+
}>;
|
|
381
|
+
/** {@link RoutingStore.cancel} — the site withdraws a job. */
|
|
382
|
+
cancel(input: {
|
|
383
|
+
jobId: string;
|
|
384
|
+
siteId: string;
|
|
385
|
+
}): Promise<boolean>;
|
|
386
|
+
/** {@link RoutingStore.cancelRequests} — cancelled jobs this runner holds. */
|
|
387
|
+
cancelRequests(runnerId: string): Promise<Grant[]>;
|
|
388
|
+
/** {@link RoutingStore.renewLeases} — extend what is still held, name what is not. */
|
|
389
|
+
renewLeases(input: {
|
|
390
|
+
runnerId: string;
|
|
391
|
+
leases: readonly {
|
|
392
|
+
jobId: string;
|
|
393
|
+
leaseId: string;
|
|
394
|
+
}[];
|
|
395
|
+
leaseMs: number;
|
|
396
|
+
}): Promise<{
|
|
397
|
+
renewed: {
|
|
398
|
+
jobId: string;
|
|
399
|
+
expiresAt: number;
|
|
400
|
+
}[];
|
|
401
|
+
lost: Grant[];
|
|
402
|
+
}>;
|
|
403
|
+
seen(presence: Omit<Presence, "lastSeenAt">): Promise<Presence>;
|
|
404
|
+
presence(runnerId: string): Promise<Presence | undefined>;
|
|
405
|
+
/**
|
|
406
|
+
* Lose a record, the way a real store does.
|
|
407
|
+
*
|
|
408
|
+
* A shared store drops presence for reasons this one never will — a TTL, a
|
|
409
|
+
* reschedule, a restart — and the interesting behaviour is what the relay
|
|
410
|
+
* does next. `ValkeyRoutingStore` has carried the same helper since
|
|
411
|
+
* finding 52; this is its memory twin, so the case can be written once
|
|
412
|
+
* against the implementation that is easy to reason about.
|
|
413
|
+
*/
|
|
414
|
+
dropPresenceForTests(runnerId: string): Promise<void>;
|
|
415
|
+
everyone(): Promise<Presence[]>;
|
|
416
|
+
/**
|
|
417
|
+
* Fire whatever the clock says is due, and report it.
|
|
418
|
+
*
|
|
419
|
+
* Returns the jobs it requeued so a caller can log or surface them — a
|
|
420
|
+
* timeout that fires invisibly is indistinguishable from a job that was
|
|
421
|
+
* never claimed, and those want very different debugging.
|
|
422
|
+
*/
|
|
423
|
+
sweep(): Promise<RoutedJob[]>;
|
|
424
|
+
}
|
|
425
|
+
|
|
426
|
+
/**
|
|
427
|
+
* What the relay needs from a place to keep routing state — cloud_006 §3.2.
|
|
428
|
+
*
|
|
429
|
+
* `RelayState` implements this in memory and is the reference; a Valkey-backed
|
|
430
|
+
* store implements the same thing across replicas. The interface exists so the
|
|
431
|
+
* relay depends on the *contract* rather than on either, and so the properties
|
|
432
|
+
* below are stated once rather than rediscovered per implementation.
|
|
433
|
+
*
|
|
434
|
+
* ## Every method is a decision plus its write
|
|
435
|
+
*
|
|
436
|
+
* Not one is a read the caller follows with a mutation. That is the whole
|
|
437
|
+
* design, and it is not a style preference: `claim` used to be
|
|
438
|
+
* `jobs()` → filter → mutate in the plane, which is atomic for exactly one
|
|
439
|
+
* reason — Node is single-threaded and the Maps are local. Neither survives a
|
|
440
|
+
* store on a network, and `packages/relay/test/two-replicas.test.ts` holds the
|
|
441
|
+
* resulting race as a failing assertion.
|
|
442
|
+
*
|
|
443
|
+
* **The rule for anything added here:** if a caller has to read, decide, and
|
|
444
|
+
* write back, the operation is in the wrong place. Move the decision in.
|
|
445
|
+
*
|
|
446
|
+
* ## What an implementation must guarantee
|
|
447
|
+
*
|
|
448
|
+
* 1. **`claim` is atomic.** Two callers claiming concurrently must not both
|
|
449
|
+
* receive the same job. `CLAIM_ATOMIC` is a protocol MUST.
|
|
450
|
+
* 2. **`enqueue` is idempotent by job id.** A known id returns what is already
|
|
451
|
+
* routing rather than rebuilding it — byollm_009 §4.2's replay argument
|
|
452
|
+
* rests on every write being idempotent per the instance it names.
|
|
453
|
+
* 3. **`complete` is idempotent.** A replayed result changes nothing, and the
|
|
454
|
+
* decision is made by the same operation that would have written it.
|
|
455
|
+
* 4. **Lease-scoped operations name the grant.** `takePayload`, `complete` and
|
|
456
|
+
* `releaseLeases` check the lease *id*, not just the runner — a runner
|
|
457
|
+
* survives a claim-release-reclaim cycle and a grant does not.
|
|
458
|
+
* 5. **`now()` is the only clock.** Every deadline the store stamps and every
|
|
459
|
+
* deadline it enforces come from here (§3.4). An implementation backed by a
|
|
460
|
+
* server returns that server's time, so two replicas cannot disagree about
|
|
461
|
+
* how long a lease is.
|
|
462
|
+
*
|
|
463
|
+
* ## What it must not do
|
|
464
|
+
*
|
|
465
|
+
* Hold a key, or learn about consent. `claim` takes `owners` as data because a
|
|
466
|
+
* predicate cannot travel to Valkey — and the effect is that the store cannot
|
|
467
|
+
* express an opinion about who may route, only about what it was told. That is
|
|
468
|
+
* what keeps `RELAY_BLIND` a property of the shape rather than of the code.
|
|
469
|
+
*/
|
|
470
|
+
/**
|
|
471
|
+
* One grant, named by both halves — V1-3.
|
|
472
|
+
*
|
|
473
|
+
* A job id is chosen per site, so two sites can pick the same one; the lease
|
|
474
|
+
* id is the unique thing an upstream and a daemon can both point at. Anywhere
|
|
475
|
+
* this store says "this piece of work, held by you", it says it with both.
|
|
476
|
+
*/
|
|
477
|
+
interface Grant {
|
|
478
|
+
readonly jobId: string;
|
|
479
|
+
readonly leaseId: string;
|
|
480
|
+
}
|
|
481
|
+
interface RoutingStore {
|
|
482
|
+
/** The one clock every deadline in this store is stamped from. */
|
|
483
|
+
now(): Promise<number>;
|
|
484
|
+
/**
|
|
485
|
+
* Take a stub for routing. Idempotent by id, **within a site**.
|
|
486
|
+
*
|
|
487
|
+
* A republished id belonging to the same site returns what is already
|
|
488
|
+
* routing, unchanged: a site that restarts and republishes its queue must
|
|
489
|
+
* not disturb work in flight (byollm_009 §4.2's replay argument rests on
|
|
490
|
+
* it).
|
|
491
|
+
*
|
|
492
|
+
* A republished id belonging to a **different** site is a different job —
|
|
493
|
+
* cloud_009 §3. Keys are `(site, id)`, so two sites choosing one id collide
|
|
494
|
+
* nowhere and there is nothing to refuse.
|
|
495
|
+
*
|
|
496
|
+
* cloud_008 finding 58 closed this by refusing the collision, which was as
|
|
497
|
+
* far as a single-site relay could go and left a cross-tenant existence
|
|
498
|
+
* oracle: a site knows its own stub is well-formed, so any answer it can
|
|
499
|
+
* tell apart from success confirms that somebody else holds that id. The
|
|
500
|
+
* refusal is gone with the collision.
|
|
501
|
+
*/
|
|
502
|
+
enqueue(input: {
|
|
503
|
+
id: string;
|
|
504
|
+
siteId: string;
|
|
505
|
+
stub: JobStub;
|
|
506
|
+
}): Promise<RoutedJob>;
|
|
507
|
+
/**
|
|
508
|
+
* One job, named by the site that published it — cloud_009 §3.
|
|
509
|
+
*
|
|
510
|
+
* The site id is a parameter rather than a scan, and the difference is
|
|
511
|
+
* finding eleven wearing a different hat: a reader that finds a job by id
|
|
512
|
+
* alone is a reader that can see every tenant's state through one door,
|
|
513
|
+
* which is exactly the anonymous read the debug page had and lost. The
|
|
514
|
+
* debug page is per-site or it is nothing.
|
|
515
|
+
*/
|
|
516
|
+
job(siteId: string, jobId: string): Promise<RoutedJob | undefined>;
|
|
517
|
+
jobs(): Promise<RoutedJob[]>;
|
|
518
|
+
/** Jobs a site must seal for, right now. */
|
|
519
|
+
awaiting(siteId: string): Promise<RoutedJob[]>;
|
|
520
|
+
/** Sealed results waiting to go home. */
|
|
521
|
+
finished(siteId: string): Promise<RoutedJob[]>;
|
|
522
|
+
/** Grant work to a device — one operation, because it has to be. */
|
|
523
|
+
claim(input: ClaimInput): Promise<ClaimedStub[]>;
|
|
524
|
+
/** Hand the sealed payload to the device that holds the lease. */
|
|
525
|
+
takePayload(input: {
|
|
526
|
+
jobId: string;
|
|
527
|
+
runnerId: string;
|
|
528
|
+
leaseId: string;
|
|
529
|
+
}): Promise<{
|
|
530
|
+
envelope: SealedEnvelope;
|
|
531
|
+
} | {
|
|
532
|
+
refused: HolderRefusal;
|
|
533
|
+
}>;
|
|
534
|
+
/** Record a finished job. Idempotent. */
|
|
535
|
+
complete(input: {
|
|
536
|
+
jobId: string;
|
|
537
|
+
runnerId: string;
|
|
538
|
+
/**
|
|
539
|
+
* The grant the result was produced under — cloud_008 §1.4a.
|
|
540
|
+
*
|
|
541
|
+
* `takePayload` and `releaseLeases` have always checked this and said why;
|
|
542
|
+
* the operation that *writes* checked only the runner, which survives a
|
|
543
|
+
* claim-release-reclaim cycle.
|
|
544
|
+
*/
|
|
545
|
+
leaseId: string;
|
|
546
|
+
envelope: SealedEnvelope;
|
|
547
|
+
disposition: "ok" | "error" | "canceled";
|
|
548
|
+
}): Promise<{
|
|
549
|
+
accepted: boolean;
|
|
550
|
+
duplicate?: boolean;
|
|
551
|
+
state: RoutedState;
|
|
552
|
+
} | {
|
|
553
|
+
refused: HolderRefusal;
|
|
554
|
+
}>;
|
|
555
|
+
/** Give back the grants this runner names. */
|
|
556
|
+
releaseLeases(input: {
|
|
557
|
+
runnerId: string;
|
|
558
|
+
leases: readonly {
|
|
559
|
+
jobId: string;
|
|
560
|
+
leaseId: string;
|
|
561
|
+
}[];
|
|
562
|
+
/**
|
|
563
|
+
* Why — cloud_008 §2.1. `refused` MUST be remembered and that job MUST
|
|
564
|
+
* NOT be offered to that runner again (`REFUSAL_NOT_REOFFERED`); every
|
|
565
|
+
* other reason means "not now" and must leave the job claimable by the
|
|
566
|
+
* same device, or a restart would strand its own work.
|
|
567
|
+
*/
|
|
568
|
+
reason?: ReleaseReason;
|
|
569
|
+
/**
|
|
570
|
+
* Epoch ms before which this job MUST NOT be offered to this runner
|
|
571
|
+
* again — the rate a transient refusal needs.
|
|
572
|
+
*
|
|
573
|
+
* "Not now" and "not ever" were the only two things a release could say,
|
|
574
|
+
* and a control plane declining a job for something the world can change
|
|
575
|
+
* means neither. Left immediate, the device re-claims at once and is
|
|
576
|
+
* declined again: a spin, and in a deployment one database read per turn
|
|
577
|
+
* of it.
|
|
578
|
+
*
|
|
579
|
+
* A store MUST NOT offer the job to that runner before this moment, and
|
|
580
|
+
* MUST go on offering it to every other device immediately — the whole
|
|
581
|
+
* reason it is not a refusal is that another machine may be the right
|
|
582
|
+
* one.
|
|
583
|
+
*/
|
|
584
|
+
retryAfter?: number;
|
|
585
|
+
}): Promise<string[]>;
|
|
586
|
+
/** Take a site's sealed payload for a claimed job. */
|
|
587
|
+
seal(input: {
|
|
588
|
+
jobId: string;
|
|
589
|
+
siteId: string;
|
|
590
|
+
envelope: SealedEnvelope;
|
|
591
|
+
}): Promise<{
|
|
592
|
+
state: RoutedState;
|
|
593
|
+
} | {
|
|
594
|
+
refused: "not-found" | "too-late";
|
|
595
|
+
was?: RoutedState;
|
|
596
|
+
}>;
|
|
597
|
+
/**
|
|
598
|
+
* Extend the leases this runner still holds, and name the ones it does not.
|
|
599
|
+
*
|
|
600
|
+
* One call because it is one question — cloud_008 §0.6. This was
|
|
601
|
+
* `lostLeases`, which answered only the second half, and the relay's
|
|
602
|
+
* heartbeat answered the first half with the literal `leases: []`. A daemon
|
|
603
|
+
* was therefore told, every few seconds, that none of its work had been
|
|
604
|
+
* renewed; the sweep requeued at `leaseMs` regardless of how alive the
|
|
605
|
+
* device was, and any job that took longer than a lease was handed to
|
|
606
|
+
* somebody else while the first device was still running it. The direct
|
|
607
|
+
* plane has always renewed here (`handlers.ts` §3), so this was also the two
|
|
608
|
+
* upstreams disagreeing about a rule the daemon cannot see.
|
|
609
|
+
*
|
|
610
|
+
* Renewal and loss come from one read of one state: asked separately they
|
|
611
|
+
* are two answers to "who holds this now", and under two replicas they can
|
|
612
|
+
* differ.
|
|
613
|
+
*/
|
|
614
|
+
renewLeases(input: {
|
|
615
|
+
runnerId: string;
|
|
616
|
+
leases: readonly {
|
|
617
|
+
jobId: string;
|
|
618
|
+
leaseId: string;
|
|
619
|
+
}[];
|
|
620
|
+
leaseMs: number;
|
|
621
|
+
}): Promise<{
|
|
622
|
+
renewed: readonly {
|
|
623
|
+
jobId: string;
|
|
624
|
+
expiresAt: number;
|
|
625
|
+
}[];
|
|
626
|
+
lost: readonly Grant[];
|
|
627
|
+
}>;
|
|
628
|
+
/**
|
|
629
|
+
* The site withdraws a job — cloud_008 §2.2.
|
|
630
|
+
*
|
|
631
|
+
* Returns false when the job is not this site's, which is the same scoping
|
|
632
|
+
* every other site-plane operation carries: a site must not cancel
|
|
633
|
+
* somebody else's work by guessing an id.
|
|
634
|
+
*/
|
|
635
|
+
cancel(input: {
|
|
636
|
+
jobId: string;
|
|
637
|
+
siteId: string;
|
|
638
|
+
}): Promise<boolean>;
|
|
639
|
+
/**
|
|
640
|
+
* Cancelled jobs this runner is holding, for the heartbeat to report.
|
|
641
|
+
*
|
|
642
|
+
* The relay answered `cancel: []` unconditionally, so a site could not stop
|
|
643
|
+
* a job it had already withdrawn — a device went on running work whose
|
|
644
|
+
* result nobody would accept, on somebody's own machine, at their expense.
|
|
645
|
+
*/
|
|
646
|
+
cancelRequests(runnerId: string): Promise<Grant[]>;
|
|
647
|
+
/** Record a device as present. The store stamps when. */
|
|
648
|
+
seen(presence: Omit<Presence, "lastSeenAt">): Promise<Presence>;
|
|
649
|
+
presence(runnerId: string): Promise<Presence | undefined>;
|
|
650
|
+
everyone(): Promise<Presence[]>;
|
|
651
|
+
/** Fire whatever the clock says is due, and report it. */
|
|
652
|
+
sweep(): Promise<RoutedJob[]>;
|
|
653
|
+
}
|
|
654
|
+
|
|
655
|
+
export { AWAITING_PAYLOAD_MS as A, type ClaimInput as C, type Grant as G, type HolderRefusal as H, type Presence as P, type RoutingStore as R, RelayState as a, type ReleaseReason as b, type RoutedJob as c, type RoutedState as d, routeKey as r };
|