@nimbus-sh/fabric 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +487 -0
  3. package/dist/alarms.d.ts +134 -0
  4. package/dist/alarms.d.ts.map +1 -0
  5. package/dist/alarms.js +214 -0
  6. package/dist/bindings.d.ts +316 -0
  7. package/dist/bindings.d.ts.map +1 -0
  8. package/dist/bindings.js +678 -0
  9. package/dist/ctx-exports.d.ts +47 -0
  10. package/dist/ctx-exports.d.ts.map +1 -0
  11. package/dist/ctx-exports.js +54 -0
  12. package/dist/facet-image-store.d.ts +112 -0
  13. package/dist/facet-image-store.d.ts.map +1 -0
  14. package/dist/facet-image-store.js +181 -0
  15. package/dist/fanout-pool.d.ts +223 -0
  16. package/dist/fanout-pool.d.ts.map +1 -0
  17. package/dist/fanout-pool.js +368 -0
  18. package/dist/index.d.ts +26 -0
  19. package/dist/index.d.ts.map +1 -0
  20. package/dist/index.js +25 -0
  21. package/dist/inner-do-registry.d.ts +41 -0
  22. package/dist/inner-do-registry.d.ts.map +1 -0
  23. package/dist/inner-do-registry.js +51 -0
  24. package/dist/launch-journal.d.ts +170 -0
  25. package/dist/launch-journal.d.ts.map +1 -0
  26. package/dist/launch-journal.js +154 -0
  27. package/dist/launch-pacer.d.ts +173 -0
  28. package/dist/launch-pacer.d.ts.map +1 -0
  29. package/dist/launch-pacer.js +193 -0
  30. package/dist/loader-ledger.d.ts +57 -0
  31. package/dist/loader-ledger.d.ts.map +1 -0
  32. package/dist/loader-ledger.js +91 -0
  33. package/dist/loader-pool.d.ts +315 -0
  34. package/dist/loader-pool.d.ts.map +1 -0
  35. package/dist/loader-pool.js +666 -0
  36. package/dist/process-fabric.d.ts +524 -0
  37. package/dist/process-fabric.d.ts.map +1 -0
  38. package/dist/process-fabric.js +388 -0
  39. package/dist/process-host.d.ts +132 -0
  40. package/dist/process-host.d.ts.map +1 -0
  41. package/dist/process-host.js +444 -0
  42. package/dist/vendor/errors.d.ts +24 -0
  43. package/dist/vendor/errors.d.ts.map +1 -0
  44. package/dist/vendor/errors.js +46 -0
  45. package/dist/vendor/serialize.d.ts +3 -0
  46. package/dist/vendor/serialize.d.ts.map +1 -0
  47. package/dist/vendor/serialize.js +25 -0
  48. package/dist/vendor/types.d.ts +69 -0
  49. package/dist/vendor/types.d.ts.map +1 -0
  50. package/dist/vendor/types.js +4 -0
  51. package/dist/workerd-facet-host.d.ts +207 -0
  52. package/dist/workerd-facet-host.d.ts.map +1 -0
  53. package/dist/workerd-facet-host.js +508 -0
  54. package/dist/ws-hibernation-config.d.ts +73 -0
  55. package/dist/ws-hibernation-config.d.ts.map +1 -0
  56. package/dist/ws-hibernation-config.js +93 -0
  57. package/package.json +62 -0
  58. package/src/alarms.ts +275 -0
  59. package/src/bindings.ts +871 -0
  60. package/src/ctx-exports.ts +77 -0
  61. package/src/facet-image-store.ts +196 -0
  62. package/src/fanout-pool.ts +503 -0
  63. package/src/index.ts +26 -0
  64. package/src/inner-do-registry.ts +58 -0
  65. package/src/launch-journal.ts +229 -0
  66. package/src/launch-pacer.ts +231 -0
  67. package/src/loader-ledger.ts +112 -0
  68. package/src/loader-pool.ts +984 -0
  69. package/src/process-fabric.ts +729 -0
  70. package/src/process-host.ts +566 -0
  71. package/src/vendor/errors.ts +56 -0
  72. package/src/vendor/serialize.ts +37 -0
  73. package/src/vendor/types.ts +75 -0
  74. package/src/workerd-facet-host.ts +694 -0
  75. package/src/ws-hibernation-config.ts +123 -0
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Ashish Kumar Singh and Nimbus contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in
13
+ all copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
21
+ THE SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,487 @@
1
+ # @nimbus-sh/fabric
2
+
3
+ > Part of [Nimbus](https://github.com/AshishKumar4/Nimbus), my hobby/research
4
+ > cloud OS. This README is edited and maintained with Claude (AI) and
5
+ > presented as-is.
6
+
7
+ The Cloudflare-specific half of Nimbus: the machinery for running real
8
+ programs on Durable Objects, DO facets, and the Worker Loader. Where
9
+ [`@nimbus-sh/core`](https://www.npmjs.com/package/@nimbus-sh/core) is the
10
+ backend-agnostic OS (filesystem, shell, process contracts), this package is
11
+ what that OS stands on when the host is Cloudflare — and it never imports the
12
+ OS's policy, only its shared primitives.
13
+
14
+ I extracted it because almost none of it is specific to Nimbus. Anyone who
15
+ hosts long-lived processes on Durable Objects meets the same platform
16
+ behaviors we did: `await put()` resolving before durability, one alarm per
17
+ object, a 65,536-facet lifetime budget, a frozen in-DO clock, RPC stubs that
18
+ die with their request context. This package is the machinery we built against
19
+ those behaviors, with the measured numbers that justified each mechanism
20
+ carried in the doc comments — they are the design record, and they travel with
21
+ the code on purpose.
22
+
23
+ Everything below was measured on deployed production workerd, not on
24
+ `wrangler dev` and not inferred from types, between June and August 2026.
25
+ Where a specific date matters it is given.
26
+
27
+ ## Importing it
28
+
29
+ The root export pulls `cloudflare:workers`, so `import ... from
30
+ '@nimbus-sh/fabric'` resolves only inside a Worker. Outside workerd (unit
31
+ tests, tooling) import the subpath modules directly —
32
+ `@nimbus-sh/fabric/alarms.js`, `@nimbus-sh/fabric/launch-journal.js`, and so
33
+ on. Most of the package is structurally typed against plain objects precisely
34
+ so it can be tested in bun or node.
35
+
36
+ An embedder wires three seams at composition time, each first-write-wins:
37
+
38
+ ```ts
39
+ import { setCtxExports, setSupervisorEntrypointName, setStagedBootAssembler } from '@nimbus-sh/fabric';
40
+
41
+ // In the Worker's fetch handler, once:
42
+ setCtxExports(ctx.exports);
43
+ // The name of your supervisor WorkerEntrypoint export. The fabric mints one
44
+ // binding per hosted program from it (env.SUPERVISOR inside the facet).
45
+ setSupervisorEntrypointName('MySupervisorRPC');
46
+ // Only if you use 'staged' boot specs; 'code' boots need no assembler.
47
+ setStagedBootAssembler(async (env, stage) => assembleLoaderConfig(env, stage));
48
+ ```
49
+
50
+ ## One alarm, many reasons
51
+
52
+ A Durable Object has ONE alarm, and a second `setAlarm()` silently overwrites
53
+ the first. Every alarm-driven subsystem therefore coordinates through a single
54
+ reason→deadline map in storage, with one dispatcher:
55
+
56
+ ```ts
57
+ import { DurableObject } from 'cloudflare:workers';
58
+ import { scheduleAlarm, dispatchAlarm } from '@nimbus-sh/fabric';
59
+
60
+ export class MySession extends DurableObject {
61
+ _alarmChain?: Promise<unknown>; // serializes the map's read-modify-write
62
+
63
+ async fetch(request: Request): Promise<Response> {
64
+ await scheduleAlarm(this, this.ctx, 'janitor', Date.now() + 60_000);
65
+ return new Response('ok');
66
+ }
67
+
68
+ async alarm(): Promise<void> {
69
+ await dispatchAlarm(this, this.ctx, {
70
+ janitor: async (now) => {
71
+ await this.cleanUp();
72
+ return { rearmAt: now + 60_000 }; // re-arm through the return value
73
+ },
74
+ });
75
+ }
76
+ }
77
+ ```
78
+
79
+ `scheduleAlarm` keeps the earliest deadline per reason and arms the real alarm
80
+ at the minimum across all of them. `dispatchAlarm` snapshots the fireable set
81
+ before running any handler (so a handler that re-schedules itself is not
82
+ re-fired in the same dispatch), silently drops unknown reasons (a rollback
83
+ from a deploy that added reasons must not wedge the alarm), and when no
84
+ reasons remain it deletes the map and does not re-arm — which is what lets the
85
+ object hibernate.
86
+
87
+ ## Knowing which incarnation you are
88
+
89
+ Workerd recycles isolates freely: cold starts, hibernation wakes, and resets
90
+ all hand you a fresh module scope over the same storage. The isolate
91
+ generation is a persisted counter that increments once per fresh isolate, and
92
+ process IDs derive from it (`PID_GEN_STRIDE` = 1,000,000 in core's process
93
+ table), which yields the one reset predicate everything else builds on: **a
94
+ pid at or below the current generation's base was allocated by a previous
95
+ incarnation.**
96
+
97
+ ```ts
98
+ import { maybeBumpIsolateGen } from '@nimbus-sh/fabric';
99
+
100
+ export class MySession extends DurableObject {
101
+ _isolateGen = 0;
102
+ _isolateGenPersisted = false;
103
+
104
+ async fetch(request: Request): Promise<Response> {
105
+ await maybeBumpIsolateGen(this, this.ctx); // idempotent per instance
106
+ // this._isolateGen is now this incarnation's generation
107
+ }
108
+ }
109
+ ```
110
+
111
+ The ordering inside is deliberate: adopt the persisted value first, bump only
112
+ after the `put` resolves. An unpersisted bump would be re-read by the next
113
+ boot and re-issued — two instances sharing one generation is exactly the pid
114
+ aliasing the counter exists to prevent. Note that `await put()` returning is
115
+ not durability; the output gate is what keeps a pid from generation N from
116
+ escaping before N is on disk.
117
+
118
+ ## The launch journal: surviving resets
119
+
120
+ The platform resets a Durable Object over what one turn has outstanding in
121
+ storage, and the reset destroys every write that turn had in flight. A
122
+ long-running launch holds everything in memory, so the process it is building
123
+ dies silently with the instance. The journal is what a later instance reads to
124
+ know that happened:
125
+
126
+ ```ts
127
+ import { ResidentLaunchJournal, type ResidentLaunchRecord } from '@nimbus-sh/fabric';
128
+
129
+ interface MyLaunch extends ResidentLaunchRecord {
130
+ argv: string[]; // whatever your redrive needs; the journal never reads it
131
+ }
132
+
133
+ const journal = new ResidentLaunchJournal<MyLaunch>(this.ctx.storage, {
134
+ generationBase: () => this.pidBase,
135
+ waitUntil: (p) => this.ctx.waitUntil(p),
136
+ redrive: (record, attempt) => this.launch(record.argv, attempt),
137
+ });
138
+
139
+ // Before the launch's first byte of real work:
140
+ await journal.journal({ pid, command, attempt: 0, phase: 'starting', argv });
141
+ // When the PROCESS (not the launch) ends:
142
+ await journal.release(pid);
143
+ // On the first turn after any reset (the turn pump calls this):
144
+ await journal.recoverInterrupted();
145
+ ```
146
+
147
+ Two details here cost us real incidents before they were mechanisms:
148
+
149
+ - **`put` then `sync()`.** `await storage.put()` resolves before durability.
150
+ Measured live: a launch killed in its first chunks left NO row for the
151
+ replacement instance to find, which is how the recovery this feeds sat inert
152
+ while its own test stayed green. `sync()` is the storage layer's durability
153
+ barrier; the journal writes through it on the way in and on the way out
154
+ (delete-then-sync, so a reset moments after release cannot resurrect a
155
+ process the user watched end).
156
+ - **The row lives for the process's lifetime, not the launch's.** Measured on
157
+ staging, 2026-08-13: every observed reset struck seconds AFTER the launch
158
+ settled. A launch-scoped row would already have been deleted when recovery
159
+ went looking.
160
+
161
+ Recovery applies the generation predicate (`pid <= generationBase()`), deletes
162
+ each stale row, and re-drives once per record (`RESIDENT_LAUNCH_MAX_ATTEMPT` =
163
+ 1) — a reset that recurs is not the transient kind.
164
+
165
+ ## Pacing big work across turns
166
+
167
+ One DO turn has a CPU budget of about 30 s (we were killed with `exceededCpu`
168
+ at 31.8 s and 32.5 s), and yielding inside an invocation buys nothing — CPU
169
+ accrues to the invocation, and only genuinely re-entering the object resets
170
+ it. Worse, a long turn pins the actor's only thread, so the terminal WebSocket
171
+ dies even when the work succeeds. And progress cannot be measured in
172
+ milliseconds, because the in-DO clock does not advance without I/O (0 ms
173
+ across 200,000 consecutive reads). So the pacer accounts **bytes**:
174
+
175
+ ```ts
176
+ import { LaunchPacer, LaunchTurnPump, scheduleAlarm } from '@nimbus-sh/fabric';
177
+
178
+ const pump = new LaunchTurnPump({
179
+ requestTurn: () => { void scheduleAlarm(this, this.ctx, 'launch-turn', Date.now()); },
180
+ recover: () => journal.recoverInterrupted(),
181
+ });
182
+ const pacer = new LaunchPacer(pump);
183
+
184
+ // Inside the launch, after each unit of work:
185
+ await pacer.spend(bytesJustProcessed); // suspends every LAUNCH_CHUNK_MAX_BYTES (2 MB)
186
+ // In alarm(), as one of the dispatcher's reasons:
187
+ 'launch-turn': () => pump.pump(),
188
+ ```
189
+
190
+ The pump awaits each resumed chunk, so the invocation that granted the turn is
191
+ the invocation that pays for the work — nothing runs detached in a handler's
192
+ microtask drain. A past-deadline alarm is delivered as soon as the object is
193
+ free, which makes `scheduleAlarm(..., Date.now())` a genuine "re-enter now"
194
+ primitive. Without an alarm-capable host the pump degrades to a same-context
195
+ timer: the single-turn behaviour this path always had, minus the
196
+ responsiveness.
197
+
198
+ ## Running programs: the loader pool
199
+
200
+ `LoaderPool` runs plain functions in warm dynamic-worker isolates over
201
+ `env.LOADER`. Functions are serialized with `fn.toString()`, so they must be
202
+ self-contained: no captured variables, no `this` (rejected at dispatch), and
203
+ their last parameter receives the forwarded bindings.
204
+
205
+ ```ts
206
+ import { LoaderPool } from '@nimbus-sh/fabric';
207
+
208
+ const pool = new LoaderPool(env, this.ctx, {
209
+ concurrency: 4,
210
+ tag: 'checksum',
211
+ omitSupervisor: true, // this pool needs no callback into the DO
212
+ });
213
+ try {
214
+ const sums = await pool.map(async (text: string) => {
215
+ const digest = await crypto.subtle.digest('SHA-256', new TextEncoder().encode(text));
216
+ return [...new Uint8Array(digest)].map((b) => b.toString(16).padStart(2, '0')).join('');
217
+ }, inputs);
218
+ } finally {
219
+ pool.dispose(); // releases the pool's long-lived RPC stubs
220
+ }
221
+ ```
222
+
223
+ Slots are stable (`slot = index % concurrency`) so a batch of 67 tarball
224
+ extractions reuses 4 warm isolates instead of paying 67 cold starts. Wasm
225
+ rides the loader's modules map as `{ wasm: ArrayBuffer }` — the only path that
226
+ works, since request-time `WebAssembly.compile` is CSP-blocked, RPC of a
227
+ compiled `Module` is refused by structured clone, and inlining bytes into the
228
+ module source OOMs the supervisor.
229
+
230
+ The cache key folds the function hash, the preamble hash, a wasm fingerprint,
231
+ and **the first 12 characters of the owning DO's id**. That last term is a
232
+ security lesson, not an optimization: without it, session B's pool reused
233
+ session A's warm isolate — which still carried A's `env.SUPERVISOR` binding —
234
+ and B's writes landed silently in A's filesystem while B's install reported
235
+ success. Warm isolates are scoped to one session unless a pool explicitly opts
236
+ into `cacheScope: 'global'`, which is reserved for stateless compute pools
237
+ that take no supervisor binding and retain no user state.
238
+
239
+ `FanoutPool` is the tier above: a single DO method can drive at most 4
240
+ concurrent Worker Loader fetches, so batches of fewer than 5 tasks run in the
241
+ coordinator through a `LoaderPool` and wider batches shard deterministically
242
+ across sibling DOs (up to 32, dispatched in phases of 4 to bound simultaneous
243
+ cold starts). Transient peer resets retry on a 250/750/1500 ms schedule; an
244
+ overloaded peer gets the 1/3/6 s one.
245
+
246
+ Every fabric call into the loader lands on a per-DO ledger: distinct ids ever
247
+ gotten — each permanently holds one of the ~5–6 dynamic-worker slots, because
248
+ a keyed `loader.get(id)` is never released — plus live and peak concurrent
249
+ Loader fetches, read via `loaderLedgerStats(ctx)`. A "Too many concurrent
250
+ dynamic workers" refusal classifies as `dynamic_worker_cap` and is annotated
251
+ with the ids actually holding slots. Measurement and honest failure naming
252
+ only — no admission control, because the cap is the platform's and
253
+ approximate, and a gate on an approximate number would refuse work the
254
+ platform would have run.
255
+
256
+ ## Running processes: the resident fabric
257
+
258
+ A resident process — a dev server, a socket runner, an attached TUI — is a DO
259
+ facet whose class comes from a dynamic worker. `openResidentFacet` is the one
260
+ way such a process comes into existence; `ProcessFabric` is the lifecycle
261
+ around it:
262
+
263
+ ```ts
264
+ import { ProcessFabric, createProcessHost } from '@nimbus-sh/fabric';
265
+
266
+ const fabric = new ProcessFabric(createProcessHost('facet', this.ctx, env, () => diskReader));
267
+
268
+ const handle = await fabric.startResidentProcess({
269
+ startContract: 'boot', // or 'lifetime' — see below
270
+ pid,
271
+ workerKey: `nimbus-process:${this.ctx.id}:${pid}`,
272
+ boot: { kind: 'code', code: spec }, // a ResidentCodeSpec
273
+ startArgs: { port: 3000 },
274
+ onWriterActivated: (writerId) => this.writers.add(writerId),
275
+ onWriterRetired: (writerId) => this.writers.delete(writerId),
276
+ });
277
+
278
+ const payload = await handle.booted();
279
+ // Inbound HTTP for the process's ports:
280
+ const response = await handle.routeTarget.handleHttpRequest(request);
281
+ // Teardown:
282
+ handle.kill();
283
+ await handle.done;
284
+ ```
285
+
286
+ The dynamic worker must export a Durable Object class named `NimbusProcess`
287
+ (`RESIDENT_PROCESS_CLASS`) with `startProcess(args)` and
288
+ `handleHttpRequest(request)`. Its `startProcess` declares one of two contracts:
289
+ `'lifetime'` (the call is held open for the process's whole life and settles at
290
+ exit — an attached TUI) or `'boot'` (the call returns a payload once the
291
+ process is up and the facet stays resident — a server).
292
+
293
+ Pieces worth knowing about, each earned the hard way:
294
+
295
+ - **The slot book.** A Durable Object admits 65,536 facets over its LIFETIME —
296
+ the IDs are append-only and never reclaimed, so the bound is on facets ever
297
+ created. Naming facets after pids burned one ID per spawn with no way back.
298
+ Reusing a NAME costs no new ID, so facet names come from a per-DO free list
299
+ (`proc-slot-<n>`, lowest reused first), and a slot is released only after
300
+ `facets.abort` + `facets.delete` — a slot handed out during teardown would
301
+ put two processes on one name. The names the book does mint are counted
302
+ durably — `facetIdBudget(ctx)` reports `{ consumed, budget }`, first uses
303
+ only, adopted across resets — and a creation failure with the budget
304
+ consumed names the budget and the count instead of repeating the platform's
305
+ opaque message. Exhaustion is permanent for the object, so it is the one
306
+ failure worth naming precisely.
307
+ - **At-most-once start.** The facet's start callback re-running would
308
+ re-execute the user's program, answering a request from a process the user
309
+ never started. Both re-entry cases (released, lost) throw instead.
310
+ - **Boot specs name large members by VFS path.** A whole structured-clone RPC
311
+ value caps at 32 MiB, and one node snapshot alone serialized to 44,252,709
312
+ bytes. `vfsWasmModules` and `vfsTextModules` send paths; the hosting actor
313
+ reads the bytes through the `ResidentDiskReader` it was given, inside the
314
+ loader's cache-miss callback, so they exist only for the duration of the
315
+ load. Text images are verified against the digest their own path claims —
316
+ a truncated image would otherwise boot as silently-wrong code.
317
+ - **The substrate is one deployment-wide value** (`createProcessHost`'s mode,
318
+ `'facet'` or `'peer'`), never per-spawn. No program name, mode, or payload
319
+ size reaches the choice.
320
+
321
+ What each substrate costs, measured on the production shape:
322
+
323
+ | | spawn | memory | CPU | SQLite |
324
+ |---|---|---|---|---|
325
+ | facet | 8–16 ms | independent (~208 MiB each) | SHARED | own |
326
+ | peer | 242–359 ms | independent | independent | own |
327
+
328
+ Facet CPU is shared because facets are separate isolates inside one actor
329
+ thread: awaiting I/O yields it completely, but a deliberate 9,956 ms CPU burn
330
+ stalled a sibling for 9,966 ms. A peer pays roughly 20× the spawn cost to buy
331
+ that back, and verifies its placement rather than assuming it — a module-scope
332
+ UUID token is compared across the hop, up to 4 sibling names tried, because a
333
+ peer that co-located shares the CPU it was chosen to escape.
334
+
335
+ The substrates also differ in image delivery, stated in the
336
+ `ProcessImageDelivery` contract rather than smoothed over: a facet shares its
337
+ session's Durable Object, so the session's store is reachable by
338
+ copy-on-write (`ctx.facets.clone`: 18–31 ms for a 45.73 MB corpus, 34–54 ms
339
+ for 1 GB — flat, because nothing is copied) but also shares the session's
340
+ ~10 GiB storage budget. A peer brings its own budget and no reflink: clone is
341
+ same-object-only and workerd exposes no `VACUUM INTO`, `ATTACH`, or
342
+ `sqlite3_backup` across objects. And a clone hazard we measured rather than
343
+ assumed: ANY unresolvable `src` — a typo, a name not created yet — silently
344
+ EMPTIES the destination and reports success. `cloneFacetStorage` is the one
345
+ way the fabric calls clone: it takes the caller's `populated(name)` probe and
346
+ asserts it positively on the source before the clone and on the destination
347
+ after, so a typo is refused before the platform call and a wiped destination
348
+ is never reported as success. An emptied facet still shows a 4,096-byte
349
+ database — one page — which is why the probe must find the caller's own data,
350
+ not a non-zero size.
351
+
352
+ ## The image store
353
+
354
+ `FacetImageStore` materializes generated boot images into a content-addressed
355
+ store (`var/lib/nimbus/facet-images/<sha256>.js`) through a small
356
+ `FacetImageBlobStore` port — the embedder owns the disk, the store owns the
357
+ protocol:
358
+
359
+ - **Root before the first byte.** The whole root set is registered
360
+ synchronously before any byte lands, so the sweep can never observe a
361
+ written-but-unclaimed image, however many turns the write spans.
362
+ - **Sliced writes.** One transaction takes `FACET_IMAGE_WRITE_SLICE_BYTES`
363
+ (a whole number of VFS chunks under the 1 MiB transaction bound — a slice
364
+ ending mid-chunk forces a read-back, and an oversize write falls back to
365
+ copy-on-write, which is quadratic). A 22.9 MB map written in one turn took
366
+ the session down with it about 25% of the time; sliced and paced, it
367
+ doesn't.
368
+ - **Size equality is completeness.** A write only ever grows the file from
369
+ offset zero, so an interrupted write leaves a strictly shorter file; the
370
+ reader verifies the digest before the loader sees the bytes.
371
+ - **The sweep roots off the process table.** An image is live for exactly as
372
+ long as a process boots from it. No TTL, no eviction heuristic; after a
373
+ reset the table is empty and every orphan goes.
374
+
375
+ ## Binding shims for inner workers
376
+
377
+ `NimbusLoaderRPC`, `NimbusLoadedWorker`, `NimbusLoadedEntrypoint`,
378
+ `NimbusAssetsRPC`, `NimbusDurableObjectNamespace`, and `NimbusDOStub` give a
379
+ dynamically-loaded inner Worker working `env` bindings. They exist because of
380
+ three platform behaviors, each of which cost a debugging session:
381
+
382
+ - **`WorkerStub` does not serialize**, so each hop a caller makes
383
+ (`load → getEntrypoint → fetch`) is its own `WorkerEntrypoint` class.
384
+ - **Stubs are I/O objects bound to the request that minted them** ("Cannot
385
+ perform I/O on behalf of a different request"), so the shims store CODE,
386
+ never stubs, and re-resolve through `LOADER.get(id, cb)` in the current
387
+ context — workerd caches by id, so repeated loads are close to free. The
388
+ code map is a hard-capped LRU of 32 entries: `wrangler dev`'s
389
+ rebuild-on-save loop once grew it without bound to a 128 MiB isolate crash.
390
+ - **An RPC stub's method is a wildcard property**: `method.call(ep, request)`
391
+ builds the pipelined path `method.call` and serializes `ep` as an argument,
392
+ which workerd refuses ("Entrypoints to dynamically-loaded workers cannot be
393
+ transferred"). Calls must be written `ep.method(request)`.
394
+
395
+ Nesting is capped at depth 4 (`NIMBUS_INNER_LOADER_DEPTH` raises it) —
396
+ Nimbus-in-Nimbus is fine, five levels is a runaway.
397
+
398
+ ## The platform, measured
399
+
400
+ These tables are the part of this package I most wanted to publish. They are
401
+ enforced by the code above where code can enforce them; the rest is here so
402
+ the next person does not have to measure them again. All figures are from
403
+ production workerd, June–August 2026.
404
+
405
+ ### Durable Object storage
406
+
407
+ | Invariant | Evidence |
408
+ |---|---|
409
+ | `await put()` resolves BEFORE durability; `ctx.storage.sync()` is the barrier; the output gate holds the guarantee | a launch killed in its first chunks left NO journal row (staging, 2026-08-13) |
410
+ | A reset destroys every write its turn had outstanding; an alarm write rolls back with it and the platform re-delivers the alarm to the replacement instance | the first turn after a reset is a recovery turn, for free |
411
+ | SQLite value cap is 2 MB per ROW, key length included | single-value ceiling 2,199,981 B with a 12-char key; overflow throws clean, catchable `SQLITE_TOOBIG` |
412
+ | One alarm per object; a second `setAlarm()` silently overwrites | why `ALARM_REASONS_KEY` is a map |
413
+ | Input gates stay closed across `get`/`put` | set-if-absent is atomic per DO with no CAS loop |
414
+ | A facet's own SQLite survives a fresh module scope | 7,141 rows / 45.7 MB intact across recycling — keep provenance in rows, never heap |
415
+ | ~10 GiB storage budget shared by the DO root and every facet and clone under it, with no copy-on-write credit | N clones of X bytes cost X·(N+1); crossing RESETS the object rather than raising an error |
416
+
417
+ ### Lifecycle and CPU
418
+
419
+ | Invariant | Evidence |
420
+ |---|---|
421
+ | No pending alarm ⇒ hibernation-eligible after ~10 s idle | why `dispatchAlarm` deletes the map when nothing remains |
422
+ | One-turn CPU budget ~30 s; yielding inside an invocation buys nothing; only genuine re-entry (an alarm) resets it | killed with `exceededCpu` at 31.8 s and 32.5 s |
423
+ | A long turn drops the object's WebSockets even when the work succeeds | the launch turn finished `outcome=ok` and the terminal died anyway |
424
+ | The in-DO clock does not advance without I/O | 0 ms across 200,000 consecutive `Time.now` reads — pace in bytes, hand deadlines to the host |
425
+ | Isolate generation increments on EVERY fresh isolate: cold start and hibernation wake, not only resets | `maybeBumpIsolateGen` adopts persisted truth first |
426
+ | `pid <= generation base` ⇒ previous generation | THE reset predicate; `PID_GEN_STRIDE` = 1,000,000 |
427
+ | `setTimeout`/`setInterval` prevent hibernation | one-shot self-nulling timers only |
428
+
429
+ ### Facets and dynamic workers
430
+
431
+ | Invariant | Evidence |
432
+ |---|---|
433
+ | Facet memory independent, ~208–256 MiB each; facet CPU SHARED across siblings | 9,956 ms burn stalled a sibling 9,966 ms; awaited I/O costs siblings 0 ms |
434
+ | 65,536 facets per DO LIFETIME; IDs append-only, never reclaimed; reusing a NAME costs no new ID | the slot book exists for this; `facetIdBudget` counts consumption durably and a failure at the wall names the budget |
435
+ | Dynamic-worker module map hard ceiling 67,108,864 bytes, shared across every member | 62 MiB lands, 64 MiB refused; boot cost roughly linear in map bytes and not the bottleneck (40 MiB → 1.42 s across 6,553 modules); every assembly seam refuses an over-ceiling map listing the largest members by size, because the platform's refusal names none |
436
+ | Request-time `WebAssembly.compile`/`instantiate` CSP-blocked; wasm rides the loader modules map as `{ wasm: ArrayBuffer }`, compiled at module load | RPC of a compiled `Module` refused by structured clone; inlined bytes OOMed the supervisor |
437
+ | Module scope bans I/O; `new Function` succeeds at module scope and throws at request time | code reaches a facet through the module map or not at all |
438
+ | The facet start callback fires at most once | re-running it would re-execute the user's program |
439
+ | ~5–6 concurrent dynamic workers per DO; at most 4 concurrent Loader fetches per DO method; loader-cache entries are never released | `IN_DO_THRESHOLD` = 5 sits under the fetch cap; every `loader.get(id)` permanently consumes a slot — counted per DO by the loader ledger, and a cap refusal names the ids holding them |
440
+ | `ctx.facets.clone` is same-object only, absent from `@cloudflare/workers-types` and the pinned workerd, present in production | 18–31 ms / 45.7 MB, 34–54 ms / 1 GB; an unresolvable `src` silently EMPTIES the destination and reports success — `cloneFacetStorage` enforces the both-ends validation |
441
+ | A DO dies at ~200 MiB of live wasm linear memory; reserved and written pages die at the same ceiling | lazy growth buys nothing; bound guest memory by rewriting the memory section |
442
+ | A wasm stack suspended (JSPI) in one request cannot resume in another | 3 in-context resumes took 6 ms; the first cross-context one hit a 30 s timeout |
443
+
444
+ ### RPC and stubs
445
+
446
+ | Invariant | Evidence |
447
+ |---|---|
448
+ | workerd constructs a NEW `WorkerEntrypoint` instance per RPC call | instance fields are write-only; fold state through return values |
449
+ | A stub minted in one invocation throws in another ("Cannot perform I/O on behalf of a different request") | store CODE, not stubs; re-resolve via `LOADER.get(id, cb)` — cached by id, ~free |
450
+ | `WorkerStub` does not serialize | one chained `WorkerEntrypoint` proxy class per hop |
451
+ | An RPC method is a wildcard property: `method.call(ep, r)` serializes `ep` as an argument and is refused | always `ep.method(r)` |
452
+ | Entrypoints to dynamically-loaded workers cannot transfer across Workers | HTTP travels as parts with plain `ReadableStream` bodies, re-piped through an identity stream this isolate owns |
453
+ | Structured-clone RPC cap 32 MiB | ship ≤ 28 MiB (`~6%` clone overhead); clone refusal is classified distinctly |
454
+ | RPC resources need explicit `Symbol.dispose`, including on error paths | a timeout's reject closure otherwise roots a 28 MiB payload for the full timeout |
455
+ | A binding minted BY AN ACTOR lives as long as the process; nothing holds a call open | why the facet's `SUPERVISOR` binding comes from the DO, not a stateless entrypoint |
456
+
457
+ ### WebSockets
458
+
459
+ | Invariant | Evidence |
460
+ |---|---|
461
+ | Without `setWebSocketAutoResponse(ping/pong)`, every idle-tab ping wakes the actor | ~2,880 wakes/day per idle tab; the config survives hibernation |
462
+ | A hibernatable WS owned by a DO cannot be written from a sibling `WorkerEntrypoint` isolate | sends happen in the DO's own context (relay pattern) |
463
+ | A resident process never receives a WebSocket | route targets are `handleHttpRequest` only; every socket terminates on the session DO |
464
+
465
+ ### Sharing an isolate
466
+
467
+ | Invariant | Evidence |
468
+ |---|---|
469
+ | The isolate heap ceiling is 128 MiB, and multiple DOs from one script can SHARE one isolate | resets observed below any single object's apparent usage |
470
+ | `exceededMemory` and `exceededCpu` are both uncatchable inside the dying isolate, observable only across an RPC boundary | absence of the error is not evidence of its absence |
471
+ | `process.memoryUsage()` returns 0 in DO context | any heap estimate is a lower bound; say so |
472
+
473
+ ## Relation to the other packages
474
+
475
+ `@nimbus-sh/core` is the OS this machinery hosts — fabric depends on it for
476
+ shared primitives (constants, RPC disposal, error classification) and core
477
+ never imports fabric.
478
+ [`@nimbus-sh/worker`](https://www.npmjs.com/package/@nimbus-sh/worker) is the
479
+ canonical embedder: it supplies the seams above, the supervisor entrypoint,
480
+ the session protocol, and everything user-facing. If you want the full hosted
481
+ product shape, start from `npx create-nimbus-app`; if you are building your
482
+ own thing on Durable Objects, this package and its doc comments are the part
483
+ of Nimbus you can take without taking Nimbus.
484
+
485
+ ## License
486
+
487
+ MIT.
@@ -0,0 +1,134 @@
1
+ /**
2
+ * alarms.ts — Durable Object alarm multiplexing + isolate-generation
3
+ * machinery, persisted across hibernation.
4
+ *
5
+ * Workerd hibernates Durable Objects between requests to free memory. On
6
+ * wake, the new isolate must rebuild its in-memory state from SQL — but it
7
+ * also needs to know "is this the same lifecycle as before, or did workerd
8
+ * recycle me?" That distinction matters for recovery (warmJoin vs cold init)
9
+ * and is captured by the isolate generation, a counter persisted across
10
+ * hibernations.
11
+ *
12
+ * A Durable Object has ONE alarm, and a second `setAlarm()` silently
13
+ * overwrites the first — so every alarm-driven subsystem coordinates through
14
+ * a single reason→deadline map and one dispatcher. Reasons are plain strings
15
+ * registered by the embedder: `scheduleAlarm` arms one, and `dispatchAlarm`
16
+ * runs the embedder-supplied handler for every reason whose deadline has
17
+ * passed.
18
+ */
19
+ /**
20
+ * The storage the alarm map lives in. `setAlarm` is optional because
21
+ * `wrangler dev` serves a storage without it, which is the whole reason
22
+ * scheduling degrades to a no-op instead of throwing.
23
+ */
24
+ export interface AlarmStorage {
25
+ get(key: string): Promise<unknown>;
26
+ put(key: string, value: unknown): Promise<void>;
27
+ delete(key: string): Promise<boolean>;
28
+ setAlarm?(scheduledTime: number): Promise<void>;
29
+ }
30
+ /** The hosting actor's context, as the alarm coordination reads it. */
31
+ export interface AlarmContext {
32
+ storage: AlarmStorage;
33
+ }
34
+ /**
35
+ * Multi-reason alarm coordination map.
36
+ *
37
+ * JSON-serialised `Record<reason, deadlineMsEpoch>` where keys are the
38
+ * embedder's canonical reason strings (e.g. 'w9-flush', 'log-janitor'). The
39
+ * alarm() dispatcher reads this on fire, dispatches every reason whose
40
+ * deadline has passed, and re-arms `ctx.storage.setAlarm` at the earliest
41
+ * remaining deadline.
42
+ *
43
+ * Why a map (not a single nextAlarmAt + reason): two subsystems can have
44
+ * distinct deadlines. Without the map, the later setAlarm() call would
45
+ * overwrite the earlier reason silently, breaking whichever subsystem
46
+ * expected its deadline.
47
+ *
48
+ * Forward-compat: the dispatcher silently drops unknown reasons so a
49
+ * rollback from a future deploy that added new reasons doesn't leave the
50
+ * alarm stuck.
51
+ *
52
+ * The VALUE is live production DO storage ('w1_next_alarm_reasons', from the
53
+ * workstream that introduced it) and must never change — renaming a storage
54
+ * key is a migration, and orphaned rows are the least of what it breaks.
55
+ */
56
+ export declare const ALARM_REASONS_KEY = "w1_next_alarm_reasons";
57
+ /**
58
+ * Storage key for the isolate-generation counter (cold-start +
59
+ * post-hibernation wake; one increment per fresh isolate).
60
+ *
61
+ * The VALUE is live production DO storage ('w9_isolate_gen') and must never
62
+ * change, same contract as {@link ALARM_REASONS_KEY}.
63
+ */
64
+ export declare const ISOLATE_GEN_KEY = "w9_isolate_gen";
65
+ /**
66
+ * The host instance carrying the per-instance alarm chain. The field lives on
67
+ * the embedder's DO instance so one chain serializes every alarm-map
68
+ * read-modify-write for that instance (see {@link scheduleAlarm}).
69
+ */
70
+ export interface AlarmHost {
71
+ _alarmChain?: Promise<unknown>;
72
+ }
73
+ /** The host instance carrying the isolate-generation state. */
74
+ export interface IsolateGenHost {
75
+ _isolateGen: number;
76
+ _isolateGenPersisted: boolean;
77
+ }
78
+ /**
79
+ * Schedule (or re-schedule) an alarm reason. Coordinated via a single map in
80
+ * DO storage so multiple subsystems don't clobber each other's `setAlarm()`
81
+ * calls.
82
+ *
83
+ * Semantics:
84
+ * - Reads the existing reasons map.
85
+ * - Sets `map[reason] = whenMs` IF `whenMs` is sooner than the
86
+ * currently-pending deadline for that reason (or no entry exists).
87
+ * Later-than-pending requests are silently ignored — the existing
88
+ * alarm will fire and re-arm anyway.
89
+ * - Writes the map back and calls `ctx.storage.setAlarm(min(deadlines))`.
90
+ *
91
+ * Cost: 1 storage read + 1 storage write + 1 setAlarm per call. setAlarm
92
+ * itself is billed as 1 row written per DO pricing. At a 60s janitor
93
+ * cadence, this is ~$0.05/mo/session at scale — dwarfed by the
94
+ * hibernation duration savings.
95
+ *
96
+ * Fail-soft: any throw is swallowed with a warn. On older runtimes /
97
+ * wrangler-dev where setAlarm is unavailable, this is a no-op (the
98
+ * subsystem's in-isolate setTimeout fallback continues to work).
99
+ */
100
+ export declare function scheduleAlarm(host: AlarmHost, ctx: AlarmContext, reason: string, whenMs: number): Promise<boolean>;
101
+ /**
102
+ * What one alarm handler may return: nothing, or a deadline this reason
103
+ * re-arms itself at. Re-arming through the return value keeps the map's
104
+ * read-modify-write inside the dispatcher, where it is serialized.
105
+ */
106
+ export type AlarmHandlerResult = void | {
107
+ rearmAt: number;
108
+ };
109
+ /** The embedder's reasons, each with the handler that answers it. */
110
+ export type AlarmHandlers = Record<string, (now: number) => AlarmHandlerResult | Promise<AlarmHandlerResult>>;
111
+ /**
112
+ * Multi-reason alarm dispatcher. Called from the DO's `alarm()` handler with
113
+ * the embedder's handler map.
114
+ *
115
+ * For each pending reason whose deadline has passed, run its handler.
116
+ * Handlers are awaited in place: the alarm invocation is the fresh turn a
117
+ * re-entering subsystem asked for, and it has to stay the one paying for the
118
+ * work it just released.
119
+ *
120
+ * After running fireable reasons, re-arms `ctx.storage.setAlarm` at the
121
+ * earliest remaining deadline. If no reasons remain, deletes the map key and
122
+ * does NOT call setAlarm — the DO becomes hibernation-eligible after the 10s
123
+ * idle window.
124
+ *
125
+ * Forward/back-compat: unknown reasons silently dropped. `onLegacyAlarm`
126
+ * covers an alarm that fires with no map at all — a deploy from before the
127
+ * map existed left a bare `setAlarm` behind, and the embedder decides what
128
+ * that one-time fire means (one dispatch later the map is populated by the
129
+ * next scheduleAlarm call).
130
+ */
131
+ export declare function dispatchAlarm(host: AlarmHost, ctx: AlarmContext, handlers: AlarmHandlers, onLegacyAlarm?: () => void): Promise<void>;
132
+ /** Increment + persist the isolate-gen counter once per fresh isolate. */
133
+ export declare function maybeBumpIsolateGen(host: IsolateGenHost, ctx: AlarmContext): Promise<void>;
134
+ //# sourceMappingURL=alarms.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"alarms.d.ts","sourceRoot":"","sources":["../src/alarms.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;GAiBG;AAIH;;;;GAIG;AACH,MAAM,WAAW,YAAY;IAC3B,GAAG,CAAC,GAAG,EAAE,MAAM,GAAG,OAAO,CAAC,OAAO,CAAC,CAAC;IACnC,GAAG,CAAC,GAAG,EAAE,MAAM,EAAE,KAAK,EAAE,OAAO,GAAG,OAAO,CAAC,IAAI,CAAC,CAAC;IAChD,MAAM,CAAC,GAAG,EAAE,MAAM,GAAG,OAAO,CAAC,OAAO,CAAC,CAAC;IACtC,QAAQ,CAAC,CAAC,aAAa,EAAE,MAAM,GAAG,OAAO,CAAC,IAAI,CAAC,CAAC;CACjD;AAED,uEAAuE;AACvE,MAAM,WAAW,YAAY;IAC3B,OAAO,EAAE,YAAY,CAAC;CACvB;AAED;;;;;;;;;;;;;;;;;;;;;GAqBG;AACH,eAAO,MAAM,iBAAiB,0BAA0B,CAAC;AAEzD;;;;;;GAMG;AACH,eAAO,MAAM,eAAe,mBAAmB,CAAC;AAEhD;;;;GAIG;AACH,MAAM,WAAW,SAAS;IACxB,WAAW,CAAC,EAAE,OAAO,CAAC,OAAO,CAAC,CAAC;CAChC;AAED,+DAA+D;AAC/D,MAAM,WAAW,cAAc;IAC7B,WAAW,EAAE,MAAM,CAAC;IACpB,oBAAoB,EAAE,OAAO,CAAC;CAC/B;AAED;;;;;;;;;;;;;;;;;;;;;GAqBG;AACH,wBAAgB,aAAa,CAC3B,IAAI,EAAE,SAAS,EACf,GAAG,EAAE,YAAY,EACjB,MAAM,EAAE,MAAM,EACd,MAAM,EAAE,MAAM,GACb,OAAO,CAAC,OAAO,CAAC,CA8BlB;AAED;;;;GAIG;AACH,MAAM,MAAM,kBAAkB,GAAG,IAAI,GAAG;IAAE,OAAO,EAAE,MAAM,CAAA;CAAE,CAAC;AAE5D,qEAAqE;AACrE,MAAM,MAAM,aAAa,GAAG,MAAM,CAChC,MAAM,EACN,CAAC,GAAG,EAAE,MAAM,KAAK,kBAAkB,GAAG,OAAO,CAAC,kBAAkB,CAAC,CAClE,CAAC;AAEF;;;;;;;;;;;;;;;;;;;GAmBG;AACH,wBAAgB,aAAa,CAC3B,IAAI,EAAE,SAAS,EACf,GAAG,EAAE,YAAY,EACjB,QAAQ,EAAE,aAAa,EACvB,aAAa,CAAC,EAAE,MAAM,IAAI,GACzB,OAAO,CAAC,IAAI,CAAC,CASf;AAwDD,0EAA0E;AAC1E,wBAAsB,mBAAmB,CAAC,IAAI,EAAE,cAAc,EAAE,GAAG,EAAE,YAAY,GAAG,OAAO,CAAC,IAAI,CAAC,CAyBhG"}