@nimbus-sh/fabric 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +487 -0
- package/dist/alarms.d.ts +134 -0
- package/dist/alarms.d.ts.map +1 -0
- package/dist/alarms.js +214 -0
- package/dist/bindings.d.ts +316 -0
- package/dist/bindings.d.ts.map +1 -0
- package/dist/bindings.js +678 -0
- package/dist/ctx-exports.d.ts +47 -0
- package/dist/ctx-exports.d.ts.map +1 -0
- package/dist/ctx-exports.js +54 -0
- package/dist/facet-image-store.d.ts +112 -0
- package/dist/facet-image-store.d.ts.map +1 -0
- package/dist/facet-image-store.js +181 -0
- package/dist/fanout-pool.d.ts +223 -0
- package/dist/fanout-pool.d.ts.map +1 -0
- package/dist/fanout-pool.js +368 -0
- package/dist/index.d.ts +26 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +25 -0
- package/dist/inner-do-registry.d.ts +41 -0
- package/dist/inner-do-registry.d.ts.map +1 -0
- package/dist/inner-do-registry.js +51 -0
- package/dist/launch-journal.d.ts +170 -0
- package/dist/launch-journal.d.ts.map +1 -0
- package/dist/launch-journal.js +154 -0
- package/dist/launch-pacer.d.ts +173 -0
- package/dist/launch-pacer.d.ts.map +1 -0
- package/dist/launch-pacer.js +193 -0
- package/dist/loader-ledger.d.ts +57 -0
- package/dist/loader-ledger.d.ts.map +1 -0
- package/dist/loader-ledger.js +91 -0
- package/dist/loader-pool.d.ts +315 -0
- package/dist/loader-pool.d.ts.map +1 -0
- package/dist/loader-pool.js +666 -0
- package/dist/process-fabric.d.ts +524 -0
- package/dist/process-fabric.d.ts.map +1 -0
- package/dist/process-fabric.js +388 -0
- package/dist/process-host.d.ts +132 -0
- package/dist/process-host.d.ts.map +1 -0
- package/dist/process-host.js +444 -0
- package/dist/vendor/errors.d.ts +24 -0
- package/dist/vendor/errors.d.ts.map +1 -0
- package/dist/vendor/errors.js +46 -0
- package/dist/vendor/serialize.d.ts +3 -0
- package/dist/vendor/serialize.d.ts.map +1 -0
- package/dist/vendor/serialize.js +25 -0
- package/dist/vendor/types.d.ts +69 -0
- package/dist/vendor/types.d.ts.map +1 -0
- package/dist/vendor/types.js +4 -0
- package/dist/workerd-facet-host.d.ts +207 -0
- package/dist/workerd-facet-host.d.ts.map +1 -0
- package/dist/workerd-facet-host.js +508 -0
- package/dist/ws-hibernation-config.d.ts +73 -0
- package/dist/ws-hibernation-config.d.ts.map +1 -0
- package/dist/ws-hibernation-config.js +93 -0
- package/package.json +62 -0
- package/src/alarms.ts +275 -0
- package/src/bindings.ts +871 -0
- package/src/ctx-exports.ts +77 -0
- package/src/facet-image-store.ts +196 -0
- package/src/fanout-pool.ts +503 -0
- package/src/index.ts +26 -0
- package/src/inner-do-registry.ts +58 -0
- package/src/launch-journal.ts +229 -0
- package/src/launch-pacer.ts +231 -0
- package/src/loader-ledger.ts +112 -0
- package/src/loader-pool.ts +984 -0
- package/src/process-fabric.ts +729 -0
- package/src/process-host.ts +566 -0
- package/src/vendor/errors.ts +56 -0
- package/src/vendor/serialize.ts +37 -0
- package/src/vendor/types.ts +75 -0
- package/src/workerd-facet-host.ts +694 -0
- package/src/ws-hibernation-config.ts +123 -0
|
@@ -0,0 +1,508 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* workerd-facet-host.ts — how a resident process is actually made, on workerd.
|
|
3
|
+
*
|
|
4
|
+
* `process-fabric.ts` says what a resident process IS, in terms no
|
|
5
|
+
* runtime owns: a boot spec, a start contract, a handle that can be routed to
|
|
6
|
+
* and released. This module is the one implementation of that on Cloudflare,
|
|
7
|
+
* and everything here is a workerd mechanism rather than a Nimbus concept —
|
|
8
|
+
* `ctx.facets`, the Worker Loader, `ctx.exports`, and the facet-index
|
|
9
|
+
* arithmetic the slot book exists to satisfy.
|
|
10
|
+
*
|
|
11
|
+
* The split is what lets the contract be read without the platform: a host
|
|
12
|
+
* that is not a Durable Object implements `ProcessHost` against the same
|
|
13
|
+
* `HostedProcess` and never imports this file.
|
|
14
|
+
*/
|
|
15
|
+
import { disposeRpcResource } from '@nimbus-sh/core/_shared/rpc-dispose.js';
|
|
16
|
+
import { getCtxExports, supervisorEntrypoint, supervisorEntrypointName, } from './ctx-exports.js';
|
|
17
|
+
import { beginLoaderFetch, recordLoaderId, withDynamicWorkerCapNamed, } from './loader-ledger.js';
|
|
18
|
+
import { RESIDENT_PROCESS_CLASS, requireStagedBootAssembler, residentLoaderConfig, } from './process-fabric.js';
|
|
19
|
+
export function getNimbusCtxExports() {
|
|
20
|
+
const ctxExports = getCtxExports();
|
|
21
|
+
if (!ctxExports || typeof ctxExports !== 'object') {
|
|
22
|
+
throw new Error('Nimbus: ctx.exports unavailable');
|
|
23
|
+
}
|
|
24
|
+
return ctxExports;
|
|
25
|
+
}
|
|
26
|
+
/**
|
|
27
|
+
* Mint a NimbusLoadedEntrypoint stub for a keyed dynamic worker. Used by the
|
|
28
|
+
* one-shot runtime paths, which run a program to completion inside a single
|
|
29
|
+
* request rather than leaving it resident: their module map is assembled in
|
|
30
|
+
* that stateless entrypoint's own isolate, never in a session DO.
|
|
31
|
+
*/
|
|
32
|
+
export async function createLoadedWorkerEntrypoint(ctxExports, supervisor, stage, name = null) {
|
|
33
|
+
if (!ctxExports.NimbusLoadedEntrypoint) {
|
|
34
|
+
throw new Error('Nimbus: ctx.exports.NimbusLoadedEntrypoint unavailable');
|
|
35
|
+
}
|
|
36
|
+
return await ctxExports.NimbusLoadedEntrypoint({
|
|
37
|
+
props: {
|
|
38
|
+
key: `nimbus-process:${supervisor.doId}:${supervisor.pid}`,
|
|
39
|
+
name,
|
|
40
|
+
depth: 0,
|
|
41
|
+
supervisor,
|
|
42
|
+
stage,
|
|
43
|
+
},
|
|
44
|
+
});
|
|
45
|
+
}
|
|
46
|
+
/**
|
|
47
|
+
* Total bytes a dynamic Worker's module map may carry, across every member of
|
|
48
|
+
* it. A hard platform limit, not a policy knob: 62 MiB lands and 64 MiB is
|
|
49
|
+
* refused with "Dynamic Worker code size (N bytes) exceeds the maximum allowed
|
|
50
|
+
* size of 67108864 bytes", confirmed at five sizes with two trials each. The
|
|
51
|
+
* budget is shared, so a ruby process is already 34.3 MiB down before its disk
|
|
52
|
+
* is counted.
|
|
53
|
+
*/
|
|
54
|
+
export const DYNAMIC_WORKER_CODE_LIMIT_BYTES = 67_108_864;
|
|
55
|
+
/**
|
|
56
|
+
* Refuse a module map over {@link DYNAMIC_WORKER_CODE_LIMIT_BYTES}, naming
|
|
57
|
+
* the largest members. The platform's own refusal reports one number for a
|
|
58
|
+
* budget shared across every member of the map, which tells the operator
|
|
59
|
+
* nothing about WHAT to shrink — so every fabric seam that assembles a map
|
|
60
|
+
* runs this before the loader sees it.
|
|
61
|
+
*
|
|
62
|
+
* Costed to its two paths. Under the ceiling: one length read per member —
|
|
63
|
+
* UTF-16 code units for text, which equal UTF-8 bytes for the ASCII module
|
|
64
|
+
* text the generators emit and undercount otherwise; the platform's own
|
|
65
|
+
* refusal still backstops the exotic case, because this check exists to name
|
|
66
|
+
* members, not to be the ceiling. Over it: exact UTF-8 sizes, computed only
|
|
67
|
+
* then, sorted so the biggest lever is first.
|
|
68
|
+
*/
|
|
69
|
+
export function assertModuleMapWithinCodeLimit(modules) {
|
|
70
|
+
let estimate = 0;
|
|
71
|
+
for (const content of Object.values(modules)) {
|
|
72
|
+
estimate += memberBytes(content, null);
|
|
73
|
+
}
|
|
74
|
+
if (estimate <= DYNAMIC_WORKER_CODE_LIMIT_BYTES)
|
|
75
|
+
return;
|
|
76
|
+
const encoder = new TextEncoder();
|
|
77
|
+
const sized = Object.entries(modules)
|
|
78
|
+
.map(([name, content]) => ({ name, bytes: memberBytes(content, encoder) }))
|
|
79
|
+
.sort((a, b) => b.bytes - a.bytes);
|
|
80
|
+
const total = sized.reduce((sum, member) => sum + member.bytes, 0);
|
|
81
|
+
const top = sized.slice(0, 5)
|
|
82
|
+
.map(({ name, bytes }) => `'${name}' (${bytes.toLocaleString('en-US')} bytes)`)
|
|
83
|
+
.join(', ');
|
|
84
|
+
throw new Error(`Nimbus: dynamic-worker module map is ${total.toLocaleString('en-US')} bytes, over the `
|
|
85
|
+
+ `${DYNAMIC_WORKER_CODE_LIMIT_BYTES.toLocaleString('en-US')}-byte platform ceiling shared by `
|
|
86
|
+
+ `every member. Largest members: ${top}`);
|
|
87
|
+
}
|
|
88
|
+
/**
|
|
89
|
+
* Bytes one module-map member carries, across the loader's content kinds
|
|
90
|
+
* (plain string, `{ js | cjs | py | text }`, `{ wasm | data }`). With an
|
|
91
|
+
* encoder, text is measured exactly; without one, by code-unit length.
|
|
92
|
+
*/
|
|
93
|
+
function memberBytes(content, encoder) {
|
|
94
|
+
const textBytes = (text) => encoder ? encoder.encode(text).byteLength : text.length;
|
|
95
|
+
if (typeof content === 'string')
|
|
96
|
+
return textBytes(content);
|
|
97
|
+
if (content !== null && typeof content === 'object') {
|
|
98
|
+
for (const value of Object.values(content)) {
|
|
99
|
+
if (typeof value === 'string')
|
|
100
|
+
return textBytes(value);
|
|
101
|
+
if (value instanceof ArrayBuffer)
|
|
102
|
+
return value.byteLength;
|
|
103
|
+
if (ArrayBuffer.isView(value))
|
|
104
|
+
return value.byteLength;
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
return 0;
|
|
108
|
+
}
|
|
109
|
+
function facetContainer(ctx) {
|
|
110
|
+
const facets = ctx.facets;
|
|
111
|
+
if (!facets || typeof facets.get !== 'function') {
|
|
112
|
+
throw new Error('Nimbus: ctx.facets is unavailable in this Durable Object; '
|
|
113
|
+
+ 'resident processes cannot be hosted');
|
|
114
|
+
}
|
|
115
|
+
return facets;
|
|
116
|
+
}
|
|
117
|
+
/**
|
|
118
|
+
* Fork one facet's entire SQLite into another by copy-on-write — the one way
|
|
119
|
+
* the fabric calls `ctx.facets.clone`, because the raw call carries a hazard
|
|
120
|
+
* measured on production workerd: ANY `src` that does not resolve to a
|
|
121
|
+
* populated facet — a typo, a name not created yet, not merely the obvious
|
|
122
|
+
* `''`/`'.'`/`'/'` — silently EMPTIES the destination and reports success. A
|
|
123
|
+
* blocklist of bad names would pass a typo straight through and wipe a
|
|
124
|
+
* process's filesystem while returning ok, so validation is positive on both
|
|
125
|
+
* ends: the source must answer as populated before the clone runs, the
|
|
126
|
+
* destination must answer as populated after it, and anything else fails
|
|
127
|
+
* loud.
|
|
128
|
+
*
|
|
129
|
+
* `populated` is the caller's probe because populated-ness is the caller's
|
|
130
|
+
* schema. The hosting actor cannot read a facet's SQLite from outside it, and
|
|
131
|
+
* an EMPTIED facet still reports a 4,096-byte database — one page, the empty
|
|
132
|
+
* file — so only a check that positively finds the caller's own data means
|
|
133
|
+
* anything. The post-clone probe is not redundant with the pre-clone one: the
|
|
134
|
+
* probe answers off the caller's accounting, and the capture races un-awaited
|
|
135
|
+
* writes to the source, so the destination is verified rather than inferred.
|
|
136
|
+
*
|
|
137
|
+
* The primitive itself, measured: a reflink, 18–31 ms for a 45.73 MB corpus
|
|
138
|
+
* and 34–54 ms for 1 GB — flat, because nothing is copied — with the data
|
|
139
|
+
* visible from the destination's constructor. Same-Durable-Object only.
|
|
140
|
+
* Quiesce and await writes to the source first; the destination name consumes
|
|
141
|
+
* a facet ID on first use like any other facet name; and the shared ~10 GiB
|
|
142
|
+
* storage budget grants no copy-on-write credit — crossing it resets the
|
|
143
|
+
* object rather than raising an error.
|
|
144
|
+
*/
|
|
145
|
+
export async function cloneFacetStorage(ctx, clone) {
|
|
146
|
+
const facets = facetContainer(ctx);
|
|
147
|
+
if (typeof facets.clone !== 'function') {
|
|
148
|
+
throw new Error('Nimbus: ctx.facets.clone is unavailable in this runtime; the reflink image '
|
|
149
|
+
+ 'path needs deployed Cloudflare workerd (local workerd <= 1.20260603.1 lacks it)');
|
|
150
|
+
}
|
|
151
|
+
const { src, dst } = clone;
|
|
152
|
+
if (!(await clone.populated(src))) {
|
|
153
|
+
throw new Error(`Nimbus: refusing to clone facet '${src}' into '${dst}': the source does not `
|
|
154
|
+
+ 'answer as populated, and cloning an unresolvable source silently EMPTIES '
|
|
155
|
+
+ 'the destination while reporting success');
|
|
156
|
+
}
|
|
157
|
+
facets.clone(src, dst);
|
|
158
|
+
if (!(await clone.populated(dst))) {
|
|
159
|
+
throw new Error(`Nimbus: clone of facet '${src}' left the destination '${dst}' without the `
|
|
160
|
+
+ "source's data; the destination must not be booted from");
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
/**
|
|
164
|
+
* The facet name for a slot. Reused, and that is the entire point.
|
|
165
|
+
*
|
|
166
|
+
* A Durable Object admits 65,536 facets over its LIFETIME: the IDs are
|
|
167
|
+
* append-only and are never reclaimed, so the bound is on facets ever CREATED,
|
|
168
|
+
* not facets alive at once. Naming a facet after its pid, when pids never
|
|
169
|
+
* repeat, therefore burned one of those IDs on every spawn — a long-lived
|
|
170
|
+
* session would eventually exhaust its facet index with no way back, and the
|
|
171
|
+
* failure is unrecoverable rather than merely slow.
|
|
172
|
+
*
|
|
173
|
+
* Reusing a NAME costs no new ID. So the name comes from a free list and the
|
|
174
|
+
* pid stays what it always was: the process identity in the ProcessTable. The
|
|
175
|
+
* two were only ever conflated because one of them happened to be handy.
|
|
176
|
+
*/
|
|
177
|
+
export function residentFacetName(slot) {
|
|
178
|
+
return `proc-slot-${slot}`;
|
|
179
|
+
}
|
|
180
|
+
/**
|
|
181
|
+
* Facet IDs a Durable Object is granted over its LIFETIME. Append-only and
|
|
182
|
+
* never reclaimed, so crossing it is unrecoverable for the object — which is
|
|
183
|
+
* why the ledger below counts consumption durably instead of leaving the
|
|
184
|
+
* bound as prose the slot book merely respects.
|
|
185
|
+
*/
|
|
186
|
+
export const FACET_ID_LIFETIME_BUDGET = 65_536;
|
|
187
|
+
/** Where the ledger persists the count of facet names ever minted. */
|
|
188
|
+
export const FACET_NAME_HIGH_WATER_KEY = 'fabric_facet_name_high_water';
|
|
189
|
+
/**
|
|
190
|
+
* Slot books, per hosting actor, because the facet index is per Durable
|
|
191
|
+
* Object.
|
|
192
|
+
*
|
|
193
|
+
* Keyed weakly off `ctx`, and that is sound rather than lossy: a facet cannot
|
|
194
|
+
* outlive the Durable Object hosting it, so a book that goes away with its
|
|
195
|
+
* host describes nothing that still exists. A fresh incarnation restarts at
|
|
196
|
+
* slot 0 and re-attaches to the SQLite a previous incarnation left there —
|
|
197
|
+
* which is safe for the reason the store is sealed until it has reconciled.
|
|
198
|
+
* Its persisted cursor is either datable against the current authority, in
|
|
199
|
+
* which case the ACQUIRE delta brings it current, or it carries a different
|
|
200
|
+
* VFS epoch, in which case `invalidatedSince` can only answer poison and the
|
|
201
|
+
* whole store is dropped. A process therefore cannot boot onto a previous
|
|
202
|
+
* tenant's filesystem even when release never ran.
|
|
203
|
+
*/
|
|
204
|
+
const slotBooks = new WeakMap();
|
|
205
|
+
function slotBook(ctx) {
|
|
206
|
+
let book = slotBooks.get(ctx);
|
|
207
|
+
if (!book) {
|
|
208
|
+
const created = {
|
|
209
|
+
free: [],
|
|
210
|
+
next: 0,
|
|
211
|
+
held: new Map(),
|
|
212
|
+
ledger: Promise.resolve(0),
|
|
213
|
+
ledgerKnown: 0,
|
|
214
|
+
};
|
|
215
|
+
created.ledger = Promise.resolve(ctx.storage.get(FACET_NAME_HIGH_WATER_KEY))
|
|
216
|
+
.then((value) => (typeof value === 'number' ? value : 0))
|
|
217
|
+
.catch(() => 0)
|
|
218
|
+
.then((adopted) => {
|
|
219
|
+
created.ledgerKnown = Math.max(created.ledgerKnown, adopted);
|
|
220
|
+
return adopted;
|
|
221
|
+
});
|
|
222
|
+
book = created;
|
|
223
|
+
slotBooks.set(ctx, book);
|
|
224
|
+
}
|
|
225
|
+
return book;
|
|
226
|
+
}
|
|
227
|
+
/**
|
|
228
|
+
* The lifetime facet-ID ledger: how many facet names this fabric has ever
|
|
229
|
+
* minted on the Durable Object, against the 65,536 the platform will ever
|
|
230
|
+
* grant it. `consumed` only ever counts FIRST uses — a reused name, in this
|
|
231
|
+
* incarnation or any earlier one, cost no new ID, which is the slot book's
|
|
232
|
+
* whole reason to exist. Surfaced so an operator can see proximity to a wall
|
|
233
|
+
* whose crossing is unrecoverable, instead of discovering it from the
|
|
234
|
+
* platform's opaque failure.
|
|
235
|
+
*/
|
|
236
|
+
export async function facetIdBudget(ctx) {
|
|
237
|
+
const book = slotBook(ctx);
|
|
238
|
+
const durable = await book.ledger;
|
|
239
|
+
return {
|
|
240
|
+
consumed: Math.max(durable, book.next),
|
|
241
|
+
budget: FACET_ID_LIFETIME_BUDGET,
|
|
242
|
+
};
|
|
243
|
+
}
|
|
244
|
+
/** Take a slot for `pid`, reusing a returned one before minting a new name. */
|
|
245
|
+
function acquireSlot(ctx, pid) {
|
|
246
|
+
const book = slotBook(ctx);
|
|
247
|
+
const existing = book.held.get(pid);
|
|
248
|
+
if (existing !== undefined)
|
|
249
|
+
return existing;
|
|
250
|
+
const reused = book.free.length > 0;
|
|
251
|
+
const slot = reused ? book.free.shift() : book.next++;
|
|
252
|
+
book.held.set(pid, slot);
|
|
253
|
+
if (!reused)
|
|
254
|
+
recordNameMinted(ctx, book);
|
|
255
|
+
return slot;
|
|
256
|
+
}
|
|
257
|
+
/**
|
|
258
|
+
* Advance the durable ledger to this incarnation's name count, if it is a new
|
|
259
|
+
* lifetime high. Chained behind adoption so the comparison is always against
|
|
260
|
+
* the real persisted value; a failed write leaves the old link's count and the
|
|
261
|
+
* next mint tries again — the ledger may transiently undercount, never over.
|
|
262
|
+
*/
|
|
263
|
+
function recordNameMinted(ctx, book) {
|
|
264
|
+
const count = book.next;
|
|
265
|
+
book.ledger = book.ledger.then(async (durable) => {
|
|
266
|
+
if (count <= durable)
|
|
267
|
+
return durable;
|
|
268
|
+
try {
|
|
269
|
+
await ctx.storage.put(FACET_NAME_HIGH_WATER_KEY, count);
|
|
270
|
+
}
|
|
271
|
+
catch {
|
|
272
|
+
return durable;
|
|
273
|
+
}
|
|
274
|
+
book.ledgerKnown = Math.max(book.ledgerKnown, count);
|
|
275
|
+
return count;
|
|
276
|
+
});
|
|
277
|
+
}
|
|
278
|
+
/**
|
|
279
|
+
* Name the facet-ID budget on a creation failure at the wall; below it, hand
|
|
280
|
+
* the error back untouched. Exhaustion is the one failure here the platform
|
|
281
|
+
* reports opaquely AND that no teardown, retry or reset can undo, so the
|
|
282
|
+
* ledger — the only witness to the real cause — does the naming. Not a
|
|
283
|
+
* threshold: the comparison is against the budget itself.
|
|
284
|
+
*/
|
|
285
|
+
function withFacetBudgetNamed(consumed, error) {
|
|
286
|
+
if (consumed < FACET_ID_LIFETIME_BUDGET)
|
|
287
|
+
return error;
|
|
288
|
+
const platform = error instanceof Error ? error.message : String(error);
|
|
289
|
+
return new Error(`Nimbus: facet creation failed with this Durable Object's `
|
|
290
|
+
+ `${FACET_ID_LIFETIME_BUDGET.toLocaleString('en-US')} facet-ID lifetime budget consumed `
|
|
291
|
+
+ `(${consumed} facet names ever created). Facet IDs are append-only and never reclaimed, `
|
|
292
|
+
+ `so this failure is permanent for the object: ${platform}`, { cause: error });
|
|
293
|
+
}
|
|
294
|
+
/** Return `pid`'s slot to the free list. */
|
|
295
|
+
function releaseSlot(ctx, pid) {
|
|
296
|
+
const book = slotBook(ctx);
|
|
297
|
+
const slot = book.held.get(pid);
|
|
298
|
+
if (slot === undefined)
|
|
299
|
+
return;
|
|
300
|
+
book.held.delete(pid);
|
|
301
|
+
book.free.push(slot);
|
|
302
|
+
book.free.sort((a, b) => a - b);
|
|
303
|
+
}
|
|
304
|
+
/**
|
|
305
|
+
* Open a resident process as a facet of the actor whose `ctx` and `env` are
|
|
306
|
+
* given, and start its runner.
|
|
307
|
+
*
|
|
308
|
+
* This is the ONE way a resident process comes into existence, and every
|
|
309
|
+
* substrate goes through it: the facet host calls it with the coordinator's
|
|
310
|
+
* own `ctx`, the peer host calls it — over one RPC — with a sibling session
|
|
311
|
+
* DO's. Everything a substrate could plausibly want to special-case is a
|
|
312
|
+
* PARAMETER here rather than a branch: which actor hosts the child, and how
|
|
313
|
+
* the boot spec's by-path members are read.
|
|
314
|
+
*/
|
|
315
|
+
export function openResidentFacet(ctx, env, disk, supervisor, params) {
|
|
316
|
+
const facets = facetContainer(ctx);
|
|
317
|
+
const book = slotBook(ctx);
|
|
318
|
+
const slot = acquireSlot(ctx, params.pid);
|
|
319
|
+
const name = residentFacetName(slot);
|
|
320
|
+
// The start callback is the ONLY way this facet is ever created, and it
|
|
321
|
+
// fires AT MOST ONCE. Every later use goes through the stub below, so the
|
|
322
|
+
// callback running a second time means the facet was released or died —
|
|
323
|
+
// and re-running it would evaluate the user's program again, answering a
|
|
324
|
+
// request from a process they never started while the one they did start
|
|
325
|
+
// is gone. Both cases are reported instead.
|
|
326
|
+
let evaluated = false;
|
|
327
|
+
let released = false;
|
|
328
|
+
const start = async () => {
|
|
329
|
+
if (released) {
|
|
330
|
+
throw new Error(`Nimbus: resident process ${params.pid} is no longer running`);
|
|
331
|
+
}
|
|
332
|
+
if (evaluated) {
|
|
333
|
+
throw new Error(`Nimbus: resident process ${params.pid} is no longer loaded (its facet was lost); `
|
|
334
|
+
+ 'it is not restarted');
|
|
335
|
+
}
|
|
336
|
+
evaluated = true;
|
|
337
|
+
return { class: residentProcessClass(ctx, env, disk, supervisor, params) };
|
|
338
|
+
};
|
|
339
|
+
let facet;
|
|
340
|
+
try {
|
|
341
|
+
facet = facets.get(name, start);
|
|
342
|
+
}
|
|
343
|
+
catch (error) {
|
|
344
|
+
releaseSlot(ctx, params.pid);
|
|
345
|
+
throw withFacetBudgetNamed(Math.max(book.ledgerKnown, book.next), error);
|
|
346
|
+
}
|
|
347
|
+
let disposed = false;
|
|
348
|
+
const release = async () => {
|
|
349
|
+
if (disposed)
|
|
350
|
+
return;
|
|
351
|
+
disposed = true;
|
|
352
|
+
released = true;
|
|
353
|
+
try {
|
|
354
|
+
facets.abort(name, new Error('Nimbus: resident process released'));
|
|
355
|
+
}
|
|
356
|
+
catch { /* already gone */ }
|
|
357
|
+
try {
|
|
358
|
+
facets.delete(name);
|
|
359
|
+
}
|
|
360
|
+
catch { /* already gone */ }
|
|
361
|
+
// Only after the facet is gone. A slot handed out while its previous
|
|
362
|
+
// tenant were still being torn down would have two processes on one name.
|
|
363
|
+
releaseSlot(ctx, params.pid);
|
|
364
|
+
};
|
|
365
|
+
let started;
|
|
366
|
+
try {
|
|
367
|
+
started = facet.startProcess(params.startArgs);
|
|
368
|
+
}
|
|
369
|
+
catch (error) {
|
|
370
|
+
void release();
|
|
371
|
+
throw withFacetBudgetNamed(Math.max(book.ledgerKnown, book.next), error);
|
|
372
|
+
}
|
|
373
|
+
// The rejection that carries the platform's failure at ID exhaustion is
|
|
374
|
+
// this one, and it is annotated AFTER awaiting the ledger — the first
|
|
375
|
+
// failure of a fresh incarnation must compare against the persisted count,
|
|
376
|
+
// not the zero its adoption read has not yet replaced.
|
|
377
|
+
started = started.catch(async (error) => {
|
|
378
|
+
const durable = await book.ledger;
|
|
379
|
+
throw withFacetBudgetNamed(Math.max(durable, book.next), error);
|
|
380
|
+
});
|
|
381
|
+
// A caller reads whichever of `started` and the lifecycle it needs, so keep
|
|
382
|
+
// the runtime from reporting the other as an unhandled rejection.
|
|
383
|
+
started.catch(() => { });
|
|
384
|
+
return {
|
|
385
|
+
started,
|
|
386
|
+
// A facet cannot die without taking its Durable Object — and this object —
|
|
387
|
+
// with it, so there is no independent death to report.
|
|
388
|
+
lost: new Promise(() => { }),
|
|
389
|
+
handleHttpRequest: (request) => facet.handleHttpRequest(request),
|
|
390
|
+
handleWebSocketRequest: (request) => facet.fetch(request),
|
|
391
|
+
release,
|
|
392
|
+
slot,
|
|
393
|
+
};
|
|
394
|
+
}
|
|
395
|
+
/**
|
|
396
|
+
* The dynamic worker's Durable Object class, minted in the caller's request
|
|
397
|
+
* context. `LOADER.get` runs its callback only on a cache miss, so a process
|
|
398
|
+
* assembles its module map at most once and the bytes never stay resident in
|
|
399
|
+
* the hosting DO's heap.
|
|
400
|
+
*/
|
|
401
|
+
function residentProcessClass(ctx, env, disk, supervisor, params) {
|
|
402
|
+
const loader = env.LOADER;
|
|
403
|
+
if (!loader || typeof loader.get !== 'function') {
|
|
404
|
+
throw new Error('Nimbus: env.LOADER binding missing or invalid. Resident processes require '
|
|
405
|
+
+ 'the Worker Loader binding; add it via worker_loaders in wrangler.jsonc.');
|
|
406
|
+
}
|
|
407
|
+
try {
|
|
408
|
+
const worker = loader
|
|
409
|
+
.get(params.workerKey, () => residentWorkerConfig(env, disk, supervisor, params.boot))
|
|
410
|
+
.getDurableObjectClass(RESIDENT_PROCESS_CLASS);
|
|
411
|
+
recordLoaderId(ctx, params.workerKey);
|
|
412
|
+
return worker;
|
|
413
|
+
}
|
|
414
|
+
catch (error) {
|
|
415
|
+
throw withDynamicWorkerCapNamed(ctx, error);
|
|
416
|
+
}
|
|
417
|
+
}
|
|
418
|
+
/**
|
|
419
|
+
* Run one program to completion as an UNKEYED dynamic worker.
|
|
420
|
+
*
|
|
421
|
+
* Unkeyed is the whole difference from `openResidentFacet`: nothing can
|
|
422
|
+
* re-resolve this worker into a later request's context, so it can never be a
|
|
423
|
+
* routeable target and never has to be released by name. It exists for the
|
|
424
|
+
* duration of one call and its stubs are dropped as that call unwinds.
|
|
425
|
+
*
|
|
426
|
+
* Shared by both substrates on purpose. `peer` places processes that have a
|
|
427
|
+
* residency to place; a one-shot has none, and shipping its fully-inline map
|
|
428
|
+
* across a sibling hop would meet the 32 MiB RPC ceiling that by-path boot
|
|
429
|
+
* specs exist to avoid — for a run that gains nothing by moving.
|
|
430
|
+
*/
|
|
431
|
+
export async function runOneShotWorker(ctx, env, supervisor, params, consume) {
|
|
432
|
+
const loader = env.LOADER;
|
|
433
|
+
if (!loader || typeof loader.load !== 'function') {
|
|
434
|
+
throw new Error('Nimbus: env.LOADER binding missing or invalid. Running a program requires '
|
|
435
|
+
+ 'the Worker Loader binding; add it via worker_loaders in wrangler.jsonc.');
|
|
436
|
+
}
|
|
437
|
+
const supervisorRpc = supervisorEntrypoint();
|
|
438
|
+
let supervisorBinding;
|
|
439
|
+
let worker;
|
|
440
|
+
let entrypoint;
|
|
441
|
+
try {
|
|
442
|
+
// Built before the capability is minted: nothing can write as this writer
|
|
443
|
+
// until there is a program to do the writing, and a map that fails to
|
|
444
|
+
// assemble should not have granted append authority on its way out.
|
|
445
|
+
let spec = await params.code();
|
|
446
|
+
assertModuleMapWithinCodeLimit(spec.modules);
|
|
447
|
+
if (supervisorRpc) {
|
|
448
|
+
params.onWriterActivated(params.writerId);
|
|
449
|
+
supervisorBinding = supervisorRpc({ props: supervisor });
|
|
450
|
+
}
|
|
451
|
+
worker = loader.load({
|
|
452
|
+
compatibilityDate: spec.compatibilityDate,
|
|
453
|
+
compatibilityFlags: spec.compatibilityFlags,
|
|
454
|
+
mainModule: spec.mainModule,
|
|
455
|
+
modules: spec.modules,
|
|
456
|
+
...(supervisorBinding ? { env: { SUPERVISOR: supervisorBinding } } : {}),
|
|
457
|
+
});
|
|
458
|
+
// The loader has taken the map; holding it here would keep a second full
|
|
459
|
+
// copy of the program alive for as long as the program runs.
|
|
460
|
+
spec = undefined;
|
|
461
|
+
entrypoint = worker.getEntrypoint();
|
|
462
|
+
// Narrowed by the runtime check; kept as a property call on the stub —
|
|
463
|
+
// extracting the method builds a pipelined `fetch.call` path workerd
|
|
464
|
+
// refuses for dynamically-loaded workers.
|
|
465
|
+
const ep = entrypoint;
|
|
466
|
+
if (typeof ep.fetch !== 'function') {
|
|
467
|
+
throw new Error('Nimbus: one-shot runtime entrypoint has no fetch method');
|
|
468
|
+
}
|
|
469
|
+
params.onLoaded?.();
|
|
470
|
+
// The unkeyed worker is a live dynamic worker for exactly this call, so
|
|
471
|
+
// the run is a Loader fetch on the hosting actor's ledger — bracketed,
|
|
472
|
+
// never wrapped: see beginLoaderFetch for the measured DO-poisoning
|
|
473
|
+
// hazard, and the pipelined-`fetch.call` note above for its sibling.
|
|
474
|
+
const endFetch = beginLoaderFetch(ctx);
|
|
475
|
+
let response;
|
|
476
|
+
try {
|
|
477
|
+
response = await ep.fetch(params.request);
|
|
478
|
+
}
|
|
479
|
+
finally {
|
|
480
|
+
endFetch();
|
|
481
|
+
}
|
|
482
|
+
try {
|
|
483
|
+
return await consume(response);
|
|
484
|
+
}
|
|
485
|
+
finally {
|
|
486
|
+
disposeRpcResource(response);
|
|
487
|
+
}
|
|
488
|
+
}
|
|
489
|
+
catch (error) {
|
|
490
|
+
throw withDynamicWorkerCapNamed(ctx, error);
|
|
491
|
+
}
|
|
492
|
+
finally {
|
|
493
|
+
disposeRpcResource(entrypoint);
|
|
494
|
+
disposeRpcResource(worker);
|
|
495
|
+
disposeRpcResource(supervisorBinding);
|
|
496
|
+
}
|
|
497
|
+
}
|
|
498
|
+
async function residentWorkerConfig(env, disk, supervisor, boot) {
|
|
499
|
+
const config = boot.kind === 'staged'
|
|
500
|
+
? await requireStagedBootAssembler()(env, boot.stage)
|
|
501
|
+
: await residentLoaderConfig(boot.code, disk());
|
|
502
|
+
assertModuleMapWithinCodeLimit(config.modules ?? {});
|
|
503
|
+
const supervisorRpc = supervisorEntrypoint();
|
|
504
|
+
if (!supervisorRpc) {
|
|
505
|
+
throw new Error(`Nimbus: ctx.exports.${supervisorEntrypointName() ?? '<supervisor entrypoint>'} unavailable`);
|
|
506
|
+
}
|
|
507
|
+
return { ...config, env: { SUPERVISOR: supervisorRpc({ props: supervisor }) } };
|
|
508
|
+
}
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ws-hibernation-config.ts — W9 (CF research §C.3 + §C.4) configuration
|
|
3
|
+
* for setWebSocketAutoResponse + setHibernatableWebSocketEventTimeout.
|
|
4
|
+
*
|
|
5
|
+
* Why a dedicated module:
|
|
6
|
+
* - NimbusSession's constructor is already crowded; the calls have
|
|
7
|
+
* specific failure modes (workerd version dependence, optional
|
|
8
|
+
* globals) that deserve isolation.
|
|
9
|
+
* - Unit-testable in Node by passing a mock ctx — the constructor
|
|
10
|
+
* itself can't be (it pulls cloudflare:workers).
|
|
11
|
+
*
|
|
12
|
+
* What this does, in order:
|
|
13
|
+
* 1. setWebSocketAutoResponse(WebSocketRequestResponsePair('ping','pong'))
|
|
14
|
+
* — vite HMR clients ping every 30s and idle xterm tabs ping per
|
|
15
|
+
* minute. Without auto-response, every ping wakes the actor from
|
|
16
|
+
* hibernation: ~2880 wakes/day per idle tab. After this, zero
|
|
17
|
+
* billable wakes for matched ping/pong frames. Auto-response config
|
|
18
|
+
* survives hibernation per the STOR/Durable Objects WebSocket
|
|
19
|
+
* Primer (wiki page id 1372566651).
|
|
20
|
+
* 2. setHibernatableWebSocketEventTimeout(5000) — bound a single
|
|
21
|
+
* hibernation message handler. Long-running work runs in facets
|
|
22
|
+
* with their own CPU budget; the supervisor's WS handlers should
|
|
23
|
+
* enqueue then return. 5 s recommended in CF research §C.3.
|
|
24
|
+
*
|
|
25
|
+
* Both calls are gated on try/catch:
|
|
26
|
+
* - workerd builds before mid-2024 don't expose either method.
|
|
27
|
+
* - WebSocketRequestResponsePair is a workerd global; absent in Node.
|
|
28
|
+
* Failure is reported honestly via the return value so /api/_diag/memory
|
|
29
|
+
* can show whether the runtime supported the configuration.
|
|
30
|
+
*/
|
|
31
|
+
/**
|
|
32
|
+
* Recommended hibernation event timeout (ms). 5 s per CF research §C.3
|
|
33
|
+
* — long enough for the heaviest non-facet WS message handler observed
|
|
34
|
+
* in W5 telemetry (~120 ms p99 for a heavy autocomplete request), short
|
|
35
|
+
* enough to bound a runaway handler before it pins the actor.
|
|
36
|
+
*/
|
|
37
|
+
export declare const NIMBUS_HIBERNATION_EVENT_TIMEOUT_MS = 5000;
|
|
38
|
+
/** Public ping/pong contract — clients send `ping`, receive `pong`. */
|
|
39
|
+
export declare const WS_AUTO_RESPONSE_REQUEST = "ping";
|
|
40
|
+
export declare const WS_AUTO_RESPONSE_RESPONSE = "pong";
|
|
41
|
+
/**
|
|
42
|
+
* The hibernation controls this module configures. Both are optional because
|
|
43
|
+
* both are version-dependent: a workerd that predates one may still expose the
|
|
44
|
+
* other, and neither exists in Node.
|
|
45
|
+
*/
|
|
46
|
+
export interface WsHibernationHost {
|
|
47
|
+
setWebSocketAutoResponse?(pair: WebSocketRequestResponsePair): void;
|
|
48
|
+
setHibernatableWebSocketEventTimeout?(timeoutMs: number): void;
|
|
49
|
+
}
|
|
50
|
+
export interface WsHibernationConfigResult {
|
|
51
|
+
/** True iff `setWebSocketAutoResponse` ran without throwing. */
|
|
52
|
+
autoResponseConfigured: boolean;
|
|
53
|
+
/**
|
|
54
|
+
* The timeout (in ms) we successfully set, or null if the call wasn't
|
|
55
|
+
* available / threw. Reported separately from autoResponseConfigured
|
|
56
|
+
* so a partial-support workerd shows the partial truth.
|
|
57
|
+
*/
|
|
58
|
+
timeoutSetMs: number | null;
|
|
59
|
+
/** Optional error message — human-readable, never thrown. */
|
|
60
|
+
autoResponseError?: string;
|
|
61
|
+
timeoutError?: string;
|
|
62
|
+
}
|
|
63
|
+
/**
|
|
64
|
+
* Configure WS hibernation behaviours on a DurableObjectState. Idempotent
|
|
65
|
+
* — safe to call multiple times. Returns a result that NimbusSession
|
|
66
|
+
* surfaces via /api/_diag/memory under the `hib` key.
|
|
67
|
+
*
|
|
68
|
+
* The `ctx` parameter is structurally typed (anything with the right
|
|
69
|
+
* methods works) so this module stays Node-testable. In production the
|
|
70
|
+
* caller passes the real `this.ctx` from NimbusSession's constructor.
|
|
71
|
+
*/
|
|
72
|
+
export declare function configureWsHibernation(ctx: WsHibernationHost): WsHibernationConfigResult;
|
|
73
|
+
//# sourceMappingURL=ws-hibernation-config.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"ws-hibernation-config.d.ts","sourceRoot":"","sources":["../src/ws-hibernation-config.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA6BG;AAIH;;;;;GAKG;AACH,eAAO,MAAM,mCAAmC,OAAO,CAAC;AAExD,uEAAuE;AACvE,eAAO,MAAM,wBAAwB,SAAS,CAAC;AAC/C,eAAO,MAAM,yBAAyB,SAAS,CAAC;AAEhD;;;;GAIG;AACH,MAAM,WAAW,iBAAiB;IAChC,wBAAwB,CAAC,CAAC,IAAI,EAAE,4BAA4B,GAAG,IAAI,CAAC;IACpE,oCAAoC,CAAC,CAAC,SAAS,EAAE,MAAM,GAAG,IAAI,CAAC;CAChE;AAED,MAAM,WAAW,yBAAyB;IACxC,gEAAgE;IAChE,sBAAsB,EAAE,OAAO,CAAC;IAChC;;;;OAIG;IACH,YAAY,EAAE,MAAM,GAAG,IAAI,CAAC;IAC5B,6DAA6D;IAC7D,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAC3B,YAAY,CAAC,EAAE,MAAM,CAAC;CACvB;AAED;;;;;;;;GAQG;AACH,wBAAgB,sBAAsB,CACpC,GAAG,EAAE,iBAAiB,GACrB,yBAAyB,CA0C3B"}
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ws-hibernation-config.ts — W9 (CF research §C.3 + §C.4) configuration
|
|
3
|
+
* for setWebSocketAutoResponse + setHibernatableWebSocketEventTimeout.
|
|
4
|
+
*
|
|
5
|
+
* Why a dedicated module:
|
|
6
|
+
* - NimbusSession's constructor is already crowded; the calls have
|
|
7
|
+
* specific failure modes (workerd version dependence, optional
|
|
8
|
+
* globals) that deserve isolation.
|
|
9
|
+
* - Unit-testable in Node by passing a mock ctx — the constructor
|
|
10
|
+
* itself can't be (it pulls cloudflare:workers).
|
|
11
|
+
*
|
|
12
|
+
* What this does, in order:
|
|
13
|
+
* 1. setWebSocketAutoResponse(WebSocketRequestResponsePair('ping','pong'))
|
|
14
|
+
* — vite HMR clients ping every 30s and idle xterm tabs ping per
|
|
15
|
+
* minute. Without auto-response, every ping wakes the actor from
|
|
16
|
+
* hibernation: ~2880 wakes/day per idle tab. After this, zero
|
|
17
|
+
* billable wakes for matched ping/pong frames. Auto-response config
|
|
18
|
+
* survives hibernation per the STOR/Durable Objects WebSocket
|
|
19
|
+
* Primer (wiki page id 1372566651).
|
|
20
|
+
* 2. setHibernatableWebSocketEventTimeout(5000) — bound a single
|
|
21
|
+
* hibernation message handler. Long-running work runs in facets
|
|
22
|
+
* with their own CPU budget; the supervisor's WS handlers should
|
|
23
|
+
* enqueue then return. 5 s recommended in CF research §C.3.
|
|
24
|
+
*
|
|
25
|
+
* Both calls are gated on try/catch:
|
|
26
|
+
* - workerd builds before mid-2024 don't expose either method.
|
|
27
|
+
* - WebSocketRequestResponsePair is a workerd global; absent in Node.
|
|
28
|
+
* Failure is reported honestly via the return value so /api/_diag/memory
|
|
29
|
+
* can show whether the runtime supported the configuration.
|
|
30
|
+
*/
|
|
31
|
+
import { errorText } from '@nimbus-sh/core/_shared/error-text.js';
|
|
32
|
+
/**
|
|
33
|
+
* Recommended hibernation event timeout (ms). 5 s per CF research §C.3
|
|
34
|
+
* — long enough for the heaviest non-facet WS message handler observed
|
|
35
|
+
* in W5 telemetry (~120 ms p99 for a heavy autocomplete request), short
|
|
36
|
+
* enough to bound a runaway handler before it pins the actor.
|
|
37
|
+
*/
|
|
38
|
+
export const NIMBUS_HIBERNATION_EVENT_TIMEOUT_MS = 5000;
|
|
39
|
+
/** Public ping/pong contract — clients send `ping`, receive `pong`. */
|
|
40
|
+
export const WS_AUTO_RESPONSE_REQUEST = 'ping';
|
|
41
|
+
export const WS_AUTO_RESPONSE_RESPONSE = 'pong';
|
|
42
|
+
/**
|
|
43
|
+
* Configure WS hibernation behaviours on a DurableObjectState. Idempotent
|
|
44
|
+
* — safe to call multiple times. Returns a result that NimbusSession
|
|
45
|
+
* surfaces via /api/_diag/memory under the `hib` key.
|
|
46
|
+
*
|
|
47
|
+
* The `ctx` parameter is structurally typed (anything with the right
|
|
48
|
+
* methods works) so this module stays Node-testable. In production the
|
|
49
|
+
* caller passes the real `this.ctx` from NimbusSession's constructor.
|
|
50
|
+
*/
|
|
51
|
+
export function configureWsHibernation(ctx) {
|
|
52
|
+
const result = {
|
|
53
|
+
autoResponseConfigured: false,
|
|
54
|
+
timeoutSetMs: null,
|
|
55
|
+
};
|
|
56
|
+
// Step 1: auto-response. The constructor is a `declare class` in
|
|
57
|
+
// @cloudflare/workers-types, so it is in scope as a type but not on
|
|
58
|
+
// `typeof globalThis`, and a bare reference would throw in Node.
|
|
59
|
+
const workerdGlobals = globalThis;
|
|
60
|
+
const Pair = workerdGlobals.WebSocketRequestResponsePair;
|
|
61
|
+
if (typeof ctx?.setWebSocketAutoResponse === 'function' && typeof Pair === 'function') {
|
|
62
|
+
try {
|
|
63
|
+
const pair = new Pair(WS_AUTO_RESPONSE_REQUEST, WS_AUTO_RESPONSE_RESPONSE);
|
|
64
|
+
ctx.setWebSocketAutoResponse(pair);
|
|
65
|
+
result.autoResponseConfigured = true;
|
|
66
|
+
}
|
|
67
|
+
catch (e) {
|
|
68
|
+
result.autoResponseError = errorText(e);
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
else if (typeof Pair !== 'function') {
|
|
72
|
+
result.autoResponseError = 'WebSocketRequestResponsePair global not available';
|
|
73
|
+
}
|
|
74
|
+
else {
|
|
75
|
+
result.autoResponseError = 'ctx.setWebSocketAutoResponse not available';
|
|
76
|
+
}
|
|
77
|
+
// Step 2: hibernation event timeout. Independent of auto-response —
|
|
78
|
+
// a workerd that lacks the global may still support the timeout
|
|
79
|
+
// method, and vice versa.
|
|
80
|
+
if (typeof ctx?.setHibernatableWebSocketEventTimeout === 'function') {
|
|
81
|
+
try {
|
|
82
|
+
ctx.setHibernatableWebSocketEventTimeout(NIMBUS_HIBERNATION_EVENT_TIMEOUT_MS);
|
|
83
|
+
result.timeoutSetMs = NIMBUS_HIBERNATION_EVENT_TIMEOUT_MS;
|
|
84
|
+
}
|
|
85
|
+
catch (e) {
|
|
86
|
+
result.timeoutError = errorText(e);
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
else {
|
|
90
|
+
result.timeoutError = 'ctx.setHibernatableWebSocketEventTimeout not available';
|
|
91
|
+
}
|
|
92
|
+
return result;
|
|
93
|
+
}
|