@nimbus-sh/fabric 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +487 -0
  3. package/dist/alarms.d.ts +134 -0
  4. package/dist/alarms.d.ts.map +1 -0
  5. package/dist/alarms.js +214 -0
  6. package/dist/bindings.d.ts +316 -0
  7. package/dist/bindings.d.ts.map +1 -0
  8. package/dist/bindings.js +678 -0
  9. package/dist/ctx-exports.d.ts +47 -0
  10. package/dist/ctx-exports.d.ts.map +1 -0
  11. package/dist/ctx-exports.js +54 -0
  12. package/dist/facet-image-store.d.ts +112 -0
  13. package/dist/facet-image-store.d.ts.map +1 -0
  14. package/dist/facet-image-store.js +181 -0
  15. package/dist/fanout-pool.d.ts +223 -0
  16. package/dist/fanout-pool.d.ts.map +1 -0
  17. package/dist/fanout-pool.js +368 -0
  18. package/dist/index.d.ts +26 -0
  19. package/dist/index.d.ts.map +1 -0
  20. package/dist/index.js +25 -0
  21. package/dist/inner-do-registry.d.ts +41 -0
  22. package/dist/inner-do-registry.d.ts.map +1 -0
  23. package/dist/inner-do-registry.js +51 -0
  24. package/dist/launch-journal.d.ts +170 -0
  25. package/dist/launch-journal.d.ts.map +1 -0
  26. package/dist/launch-journal.js +154 -0
  27. package/dist/launch-pacer.d.ts +173 -0
  28. package/dist/launch-pacer.d.ts.map +1 -0
  29. package/dist/launch-pacer.js +193 -0
  30. package/dist/loader-ledger.d.ts +57 -0
  31. package/dist/loader-ledger.d.ts.map +1 -0
  32. package/dist/loader-ledger.js +91 -0
  33. package/dist/loader-pool.d.ts +315 -0
  34. package/dist/loader-pool.d.ts.map +1 -0
  35. package/dist/loader-pool.js +666 -0
  36. package/dist/process-fabric.d.ts +524 -0
  37. package/dist/process-fabric.d.ts.map +1 -0
  38. package/dist/process-fabric.js +388 -0
  39. package/dist/process-host.d.ts +132 -0
  40. package/dist/process-host.d.ts.map +1 -0
  41. package/dist/process-host.js +444 -0
  42. package/dist/vendor/errors.d.ts +24 -0
  43. package/dist/vendor/errors.d.ts.map +1 -0
  44. package/dist/vendor/errors.js +46 -0
  45. package/dist/vendor/serialize.d.ts +3 -0
  46. package/dist/vendor/serialize.d.ts.map +1 -0
  47. package/dist/vendor/serialize.js +25 -0
  48. package/dist/vendor/types.d.ts +69 -0
  49. package/dist/vendor/types.d.ts.map +1 -0
  50. package/dist/vendor/types.js +4 -0
  51. package/dist/workerd-facet-host.d.ts +207 -0
  52. package/dist/workerd-facet-host.d.ts.map +1 -0
  53. package/dist/workerd-facet-host.js +508 -0
  54. package/dist/ws-hibernation-config.d.ts +73 -0
  55. package/dist/ws-hibernation-config.d.ts.map +1 -0
  56. package/dist/ws-hibernation-config.js +93 -0
  57. package/package.json +62 -0
  58. package/src/alarms.ts +275 -0
  59. package/src/bindings.ts +871 -0
  60. package/src/ctx-exports.ts +77 -0
  61. package/src/facet-image-store.ts +196 -0
  62. package/src/fanout-pool.ts +503 -0
  63. package/src/index.ts +26 -0
  64. package/src/inner-do-registry.ts +58 -0
  65. package/src/launch-journal.ts +229 -0
  66. package/src/launch-pacer.ts +231 -0
  67. package/src/loader-ledger.ts +112 -0
  68. package/src/loader-pool.ts +984 -0
  69. package/src/process-fabric.ts +729 -0
  70. package/src/process-host.ts +566 -0
  71. package/src/vendor/errors.ts +56 -0
  72. package/src/vendor/serialize.ts +37 -0
  73. package/src/vendor/types.ts +75 -0
  74. package/src/workerd-facet-host.ts +694 -0
  75. package/src/ws-hibernation-config.ts +123 -0
@@ -0,0 +1,503 @@
1
+ /**
2
+ * Two-tier fan-out primitive for work that must execute in Worker Loader
3
+ * facets without tripping workerd's per-DO dynamic-worker ceiling.
4
+ *
5
+ * A single Durable Object method can drive at most four concurrent
6
+ * Worker Loader fetches before extra dispatches serialize or fail. Small
7
+ * batches therefore run in the coordinator DO through LoaderPool.
8
+ * Wider batches are sharded across sibling NimbusSession DOs, each of
9
+ * which owns its own four-loader budget.
10
+ *
11
+ * Routing is deterministic: each task has a stable key, and the key maps
12
+ * to a sibling DO shard. There is no silent fallback to width-1 execution;
13
+ * missing LOADER or NIMBUS_SESSION bindings fail loudly so install and
14
+ * runtime operations do not appear successful after partial dispatch.
15
+ */
16
+
17
+ import { serializeFunction } from './vendor/serialize.js';
18
+ import { BindingError } from './vendor/errors.js';
19
+ import { LoaderPool, type FacetTaskFn } from './loader-pool.js';
20
+ import { disposeRpcResource } from '@nimbus-sh/core/_shared/rpc-dispose.js';
21
+ import { describeError, isDoOverloaded, isTransientDoReset } from '@nimbus-sh/core/observability/oom-classify.js';
22
+ import type { WorkerLoader } from './vendor/types.js';
23
+
24
+ /**
25
+ * The sibling-session namespace the peer-DO topology routes through. Ids are
26
+ * derived from a name so a task key always lands on the same peer.
27
+ */
28
+ interface PeerSessionNamespace {
29
+ idFromName(name: string): DurableObjectId;
30
+ get(id: DurableObjectId): unknown;
31
+ }
32
+
33
+ /** The bindings a fan-out needs off the coordinator DO's env. */
34
+ export interface FanoutPoolEnv {
35
+ LOADER?: WorkerLoader;
36
+ NIMBUS_SESSION?: PeerSessionNamespace;
37
+ }
38
+
39
+ /**
40
+ * Threshold at which routing switches from coordinator-local loaders to
41
+ * sibling Durable Objects.
42
+ *
43
+ * Set to **5** so the in-DO path stays below the V8 4-loaders-per-method
44
+ * cap by construction. width < 5 stays local; width >= 5 uses sibling DOs.
45
+ */
46
+ export const IN_DO_THRESHOLD = 5;
47
+
48
+ /**
49
+ * Hard cap on concurrent peer DOs per single submitMany call. Throughput stays
50
+ * flat through this width while keeping per-request scheduler pressure bounded.
51
+ */
52
+ export const MAX_PEER_FANOUT = 32;
53
+
54
+ /**
55
+ * Bounded retries for a peer-DO shard dispatch that rejects with a
56
+ * transient platform reset (code roll-over, storage cold-start hiccup).
57
+ * Sibling DOs are addressed by stable name, so the retry re-dispatches
58
+ * the SAME shard to the re-provisioning object; the fanned-out work
59
+ * (packument resolution, tarball materialisation) is idempotent, so
60
+ * re-running a shard is safe. Budget mirrors the resolve-facet's own
61
+ * per-fetch retry policy so a single flaky cold start no longer fails a
62
+ * whole install. The same budget covers an overloaded peer, on the longer
63
+ * schedule below. Non-transient rejections (OOM, count mismatch, genuine
64
+ * task throw) are NOT retried — they propagate on the first hit.
65
+ */
66
+ export const PEER_TRANSIENT_RESET_RETRIES = 3;
67
+ export const PEER_RETRY_BACKOFF_MS = [250, 750, 1500];
68
+
69
+ /**
70
+ * Backoff for a shard whose peer DO was shed as overloaded. The object is
71
+ * alive and the shard never ran; what it needs is time for the input-gate
72
+ * queue to drain, so the schedule is an order of magnitude longer than the
73
+ * reset schedule. A whole-batch abort here used to fail an entire install.
74
+ */
75
+ export const PEER_OVERLOAD_BACKOFF_MS = [1000, 3000, 6000];
76
+
77
+ /**
78
+ * Peer shards dispatched per phase. Each phase is a barrier that costs its
79
+ * slowest member, so a wide fan-out pays ⌈shards / FANOUT_PHASE_SIZE⌉ serial
80
+ * round-trips; the size trades that serialization against simultaneous cold
81
+ * sibling DO starts.
82
+ *
83
+ * The six-barrier profile once measured on a 123-package install (21 shards of
84
+ * ~6 packages, 10.6/6.8/21.8/7.9/34.6/4.8 s) came from the shard count, not
85
+ * from this width. Capping install shards at INSTALL_PEER_CAP fixed it at the
86
+ * source and that install now clears in two phases. Widening to 8 on top of
87
+ * that bought one further barrier and doubled the simultaneous cold sibling-DO
88
+ * starts, which is the account-level pressure the phasing exists for: twelve
89
+ * concurrent Markflow installs went from 48 simultaneous peer starts to 96 and
90
+ * began timing out. Phasing does not change how many peers start, only how
91
+ * many start at once, so this width is set by the burst the scheduler
92
+ * tolerates rather than by the barrier count.
93
+ */
94
+ export const FANOUT_PHASE_SIZE = 4;
95
+
96
+ /** Argument shape for `submitMany`. */
97
+ export interface FanoutTask<A> {
98
+ /**
99
+ * Routing key for the stable-id router. Same key → same peer DO
100
+ * (when on the peer-DO path). Tests use this to predict placement.
101
+ */
102
+ key: string;
103
+ /** Argument passed to the user fn. */
104
+ args: A;
105
+ }
106
+
107
+ /** Options handed to FanoutPool's constructor. */
108
+ export interface FanoutPoolOptions {
109
+ /**
110
+ * Tag prepended to peer-DO ids and in-DO loader ids for debugging
111
+ * (e.g. "npm-install-batch"). Affects neither isolate identity (in-DO
112
+ * path uses the existing LoaderPool's tag-fold) nor peer-DO
113
+ * deterministic placement (peer ids fold tag + key).
114
+ */
115
+ tag: string;
116
+ /**
117
+ * Per-task timeout in ms. Default 60_000. Forwarded to the in-DO
118
+ * LoaderPool's submit calls and to the peer-DO RPC's own
119
+ * LoaderPool.
120
+ */
121
+ timeoutMs?: number;
122
+ /**
123
+ * Preamble bundled into every facet (in-DO and inside each peer
124
+ * DO). Same semantics as LoaderPool's preamble option.
125
+ */
126
+ preamble?: string;
127
+ /**
128
+ * Wasm modules forwarded to every facet. Same semantics as
129
+ * LoaderPool's wasmModules option.
130
+ */
131
+ wasmModules?: Record<string, ArrayBuffer>;
132
+ /**
133
+ * Extra bindings forwarded to every facet. Same semantics as
134
+ * LoaderPool's extraBindings option.
135
+ */
136
+ extraBindings?: Record<string, unknown>;
137
+ /**
138
+ * If set, skip the supervisor-RPC binding injection (mirrors
139
+ * LoaderPool's omitSupervisor flag).
140
+ */
141
+ omitSupervisor?: boolean;
142
+ /**
143
+ * Invoking process pid, baked into each facet's SUPERVISOR binding so
144
+ * filesystem RPCs (writeBatchStream) are authorized under the caller's
145
+ * credential (mirrors LoaderPool's supervisorPid). Threaded to both
146
+ * the in-DO loader pool and, via `_rpcFanoutExecute`, the peer-DO pools.
147
+ * npm install passes the shell command's `ctx.pid`; resolve leaves it 0.
148
+ */
149
+ supervisorPid?: number;
150
+ /**
151
+ * Called once per completed peer-DO dispatch phase with that phase's shard
152
+ * count and elapsed ms. Phases are barriers, so this is what tells a caller
153
+ * whether its fan-out is bounded by shard work or by the number of barriers.
154
+ * Not called on the in-DO path, which has no phases.
155
+ */
156
+ onDispatchPhase?: (width: number, elapsedMs: number) => void;
157
+ /**
158
+ * Cap on peer DOs this pool will spread one submitMany across. Defaults to
159
+ * MAX_PEER_FANOUT. Tasks beyond the cap bucket into the peers that exist and
160
+ * run through their in-peer pool, so lowering it trades peers for barriers
161
+ * without lowering total concurrency: each peer runs its bucket at
162
+ * concurrency 4, so N peers still resolve 4N tasks at once.
163
+ *
164
+ * A caller sets this when its per-task work is small enough that a peer per
165
+ * task buys nothing but round-trips — one task per peer costs ⌈tasks/
166
+ * FANOUT_PHASE_SIZE⌉ barriers, and each barrier costs a cold sibling start.
167
+ */
168
+ maxPeers?: number;
169
+ }
170
+
171
+ interface FanoutPeerStub {
172
+ _rpcFanoutExecute<R>(
173
+ fnSource: string,
174
+ args: unknown[],
175
+ poolOpts?: Record<string, unknown>,
176
+ ): Promise<{ results?: R[] }>;
177
+ }
178
+
179
+ function isFanoutPeerStub(value: unknown): value is FanoutPeerStub {
180
+ if ((typeof value !== 'object' && typeof value !== 'function') || value === null) {
181
+ return false;
182
+ }
183
+ const execute = Reflect.get(value, '_rpcFanoutExecute');
184
+ return typeof execute === 'function';
185
+ }
186
+
187
+ function fanoutPeerStub(value: unknown): FanoutPeerStub {
188
+ if ((typeof value !== 'object' && typeof value !== 'function') || value === null) {
189
+ throw new BindingError('FanoutPool: NIMBUS_SESSION.get() did not return a peer stub.');
190
+ }
191
+ if (!isFanoutPeerStub(value)) {
192
+ throw new BindingError('FanoutPool: peer stub does not expose _rpcFanoutExecute().');
193
+ }
194
+ return value;
195
+ }
196
+
197
+ /**
198
+ * Two-tier fan-out pool. Constructed by the supervisor DO; routes
199
+ * each `submitMany` call automatically based on width.
200
+ *
201
+ * Lifetime: cheap to construct (no async init). Multiple submitMany
202
+ * calls share NO state — each is dispatched fresh. The class
203
+ * exists primarily as a clean API surface; per-call dispatch state
204
+ * lives only inside submitMany's promise.
205
+ */
206
+ export class FanoutPool {
207
+ private readonly env: FanoutPoolEnv;
208
+ private readonly ctx: DurableObjectState;
209
+ private readonly opts: FanoutPoolOptions;
210
+ private readonly coordDoId: string;
211
+ private readonly coordDoIdShort: string;
212
+
213
+ constructor(rawEnv: unknown, ctx: DurableObjectState, opts: FanoutPoolOptions) {
214
+ // A host hands its whole env over; the bindings are claimed here and the
215
+ // LOADER claim is checked immediately. Hard-fail on a missing LOADER —
216
+ // LoaderPool also enforces this, but checking up front points the
217
+ // diagnostic at the fanout-pool construction site rather than the
218
+ // deferred loader-pool one.
219
+ const env = (rawEnv as FanoutPoolEnv | null | undefined) ?? {};
220
+ if (!env.LOADER || typeof env.LOADER.get !== 'function') {
221
+ throw new BindingError(
222
+ 'FanoutPool: env.LOADER binding missing or invalid. ' +
223
+ 'Add a [[worker_loaders]] entry to wrangler.jsonc.',
224
+ );
225
+ }
226
+ this.env = env;
227
+ this.ctx = ctx;
228
+ this.opts = opts;
229
+ this.coordDoId = ctx.id.toString();
230
+ this.coordDoIdShort = this.coordDoId.slice(0, 12);
231
+ }
232
+
233
+ /**
234
+ * Dispatch `tasks` across the appropriate topology and return
235
+ * results in input order.
236
+ *
237
+ * Routing:
238
+ * tasks.length < 5 -> coordinator-local LoaderPool
239
+ * tasks.length >= 5 -> sibling NimbusSession DOs
240
+ *
241
+ * Backpressure: if `tasks.length > MAX_PEER_FANOUT (32)`, tasks
242
+ * are sharded modulo `MAX_PEER_FANOUT` and each shard's bucket
243
+ * runs serially inside its assigned peer DO via the in-peer
244
+ * LoaderPool's concurrency (capped at 4 there too). A
245
+ * single submitMany call returns when ALL tasks complete (or any
246
+ * throws).
247
+ *
248
+ * `fn` is the user function executed per task. It runs INSIDE a
249
+ * Worker Loader isolate (in the in-DO path) or inside a peer DO's
250
+ * Worker Loader isolate (in the peer-DO path); same trust posture
251
+ * as LoaderPool.submit. The function is serialized via
252
+ * the vendored serializeFunction (same as LoaderPool#prepare).
253
+ */
254
+ async submitMany<A, R>(
255
+ tasks: FanoutTask<A>[],
256
+ fn: FacetTaskFn<A, R>,
257
+ ): Promise<R[]> {
258
+ if (tasks.length === 0) return [];
259
+
260
+ if (tasks.length < IN_DO_THRESHOLD) {
261
+ return this._dispatchInDo<A, R>(tasks, fn);
262
+ }
263
+ return this._dispatchPeerDo<A, R>(tasks, fn);
264
+ }
265
+
266
+ /** Report which topology a task count uses without dispatching. */
267
+ topologyFor(taskCount: number): 'in-do' | 'peer-do' | 'empty' {
268
+ if (taskCount === 0) return 'empty';
269
+ return taskCount < IN_DO_THRESHOLD ? 'in-do' : 'peer-do';
270
+ }
271
+
272
+ /**
273
+ * Compute the deterministic peer-DO id for a task key and peer count.
274
+ *
275
+ * Shape: `nbf:${tag}:${coordDoIdShort}:${shard}` where
276
+ * `shard = hash(key) mod peerCount`. Peer count is
277
+ * `min(tasks.length, MAX_PEER_FANOUT)`.
278
+ */
279
+ peerSiblingId(key: string, peerCount: number): string {
280
+ const shard = hashKeyToShard(key, peerCount);
281
+ return `nbf:${this.opts.tag}:${this.coordDoIdShort}:${shard}`;
282
+ }
283
+
284
+ // ── Private: in-DO dispatch (in-DO fanout) ──────────────────────────────
285
+
286
+ private async _dispatchInDo<A, R>(
287
+ tasks: FanoutTask<A>[],
288
+ fn: FacetTaskFn<A, R>,
289
+ ): Promise<R[]> {
290
+ // Use the existing LoaderPool. Concurrency = task count
291
+ // (capped at 4 by constructor — tasks.length is already < 5
292
+ // here, so the cap won't bite). Each task = one pool.submit;
293
+ // pool.map runs them with stable-slot reuse.
294
+ const concurrency = Math.min(tasks.length, IN_DO_THRESHOLD - 1);
295
+ const pool = new LoaderPool(this.env, this.ctx, {
296
+ concurrency,
297
+ timeoutMs: this.opts.timeoutMs,
298
+ tag: this.opts.tag,
299
+ preamble: this.opts.preamble,
300
+ wasmModules: this.opts.wasmModules,
301
+ extraBindings: this.opts.extraBindings,
302
+ omitSupervisor: this.opts.omitSupervisor,
303
+ supervisorPid: this.opts.supervisorPid,
304
+ });
305
+ try {
306
+ // pool.map runs the function over `items` with concurrency-bounded
307
+ // slot reuse. Each slot is one warm loader isolate; we get exactly
308
+ // `concurrency` loader isolates total — well under the 4-cap.
309
+ const items = tasks.map((t) => t.args);
310
+ const results = await pool.map<A, R>(fn, items);
311
+ // pool.map returns Array<R | null> (null on per-item failure with
312
+ // onError='null'/'skip'). Default onError='throw' rejects on
313
+ // first failure, so successful settle here implies all R values.
314
+ return results as R[];
315
+ } finally {
316
+ try { pool.dispose(); } catch { /* best-effort */ }
317
+ }
318
+ }
319
+
320
+ // ── Private: peer-DO dispatch (peer-DO fanout) ────────────────────────────
321
+
322
+ private async _dispatchPeerDo<A, R>(
323
+ tasks: FanoutTask<A>[],
324
+ fn: FacetTaskFn<A, R>,
325
+ ): Promise<R[]> {
326
+ const ns = this.env?.NIMBUS_SESSION;
327
+ if (!ns || typeof ns.idFromName !== 'function' || typeof ns.get !== 'function') {
328
+ throw new BindingError(
329
+ 'FanoutPool: env.NIMBUS_SESSION binding missing or invalid. ' +
330
+ 'The peer-DO topology requires it. ' +
331
+ 'Add the binding via durable_objects.bindings in wrangler.jsonc.',
332
+ );
333
+ }
334
+
335
+ // Serialize the user function ONCE here on the supervisor side.
336
+ // Each peer DO receives the same fnSource string; warm peer
337
+ // loader isolates (keyed on fnHash) reuse across calls with
338
+ // identical fns.
339
+ const fnSource = serializeFunction(fn);
340
+
341
+ // Cap peer count at MAX_PEER_FANOUT. Tasks beyond N=32 are
342
+ // bucketed into existing shards — each shard's peer DO then
343
+ // runs its bucket through its in-DO LoaderPool.map
344
+ // (concurrency capped at 4 there).
345
+ const peerCount = Math.min(tasks.length, this.opts.maxPeers ?? MAX_PEER_FANOUT);
346
+ // Group tasks by deterministic shard. Same key → same shard, so
347
+ // tests can predict which peer handles which task.
348
+ const shards = new Map<number, FanoutTask<A>[]>();
349
+ for (const t of tasks) {
350
+ const shard = hashKeyToShard(t.key, peerCount);
351
+ let bucket = shards.get(shard);
352
+ if (!bucket) {
353
+ bucket = [];
354
+ shards.set(shard, bucket);
355
+ }
356
+ bucket.push(t);
357
+ }
358
+
359
+ // Dispatch each shard to its peer DO. Build a map from
360
+ // task → its place in the original tasks array so we can
361
+ // reassemble results in input order.
362
+ const taskIndex = new Map<FanoutTask<A>, number>();
363
+ tasks.forEach((t, i) => {
364
+ taskIndex.set(t, i);
365
+ });
366
+ const results: R[] = new Array(tasks.length);
367
+
368
+ // Build one async dispatcher per shard (closure capturing siblingName,
369
+ // bucket). NOT eagerly-started — wrapped in a thunk so we can stagger
370
+ // dispatch via Promise chains without forcing all shards to start
371
+ // simultaneously.
372
+ const dispatchers: (() => Promise<void>)[] = [];
373
+ for (const [shard, bucket] of shards) {
374
+ const siblingName = `nbf:${this.opts.tag}:${this.coordDoIdShort}:${shard}`;
375
+ const id = ns.idFromName(siblingName);
376
+ const peerArgs = bucket.map((t) => t.args);
377
+ dispatchers.push(async () => {
378
+ for (let attempt = 0; ; attempt++) {
379
+ // Fresh stub per attempt: after a transient reset the previous
380
+ // stub points at a torn-down object, so a retry re-resolves the
381
+ // sibling by its stable id.
382
+ const peerStub = ns.get(id);
383
+ const stub = fanoutPeerStub(peerStub);
384
+ try {
385
+ // Each peer DO RPC call uses ONE LOADER worker on its side.
386
+ // Supervisor → peer DO is a stub.fetch / RPC method call,
387
+ // NOT an env.LOADER.get(); that's the cap-sidestep that
388
+ // makes peer-DO fanout work.
389
+ const rpcResp = await stub._rpcFanoutExecute<R>(
390
+ fnSource,
391
+ peerArgs,
392
+ {
393
+ tag: this.opts.tag,
394
+ timeoutMs: this.opts.timeoutMs,
395
+ preamble: this.opts.preamble,
396
+ wasmModules: this.opts.wasmModules,
397
+ extraBindings: this.opts.extraBindings,
398
+ omitSupervisor: this.opts.omitSupervisor,
399
+ // INSTALL-HONESTY: forward the COORDINATOR's full doId so
400
+ // the peer's LoaderPool can mint a SUPERVISOR
401
+ // binding that routes back HERE (the user's session DO),
402
+ // not to the peer DO itself. Without this, peer DOs'
403
+ // env.SUPERVISOR.writeBatch / writeBatchStream / stdout /
404
+ // ... write into the peer's own VFS — invisible to the
405
+ // user. See INSTALL-HONESTY-retro.md.
406
+ coordinatorDoId: this.coordDoId,
407
+ // Credential source for peer-side writeBatchStream — the
408
+ // invoking process pid, so package writes are authorized
409
+ // as the user (not rejected as pid:0).
410
+ supervisorPid: this.opts.supervisorPid,
411
+ },
412
+ );
413
+ try {
414
+ const peerResults = rpcResp.results ?? [];
415
+ if (peerResults.length !== bucket.length) {
416
+ throw new Error(
417
+ `peer DO returned ${peerResults.length} results for ${bucket.length} tasks ` +
418
+ `(siblingName=${siblingName})`,
419
+ );
420
+ }
421
+ // Place each result back into its original input slot.
422
+ for (let i = 0; i < bucket.length; i++) {
423
+ const origIdx = taskIndex.get(bucket[i]);
424
+ if (origIdx === undefined) {
425
+ throw new Error(`peer DO result had no original task index (siblingName=${siblingName})`);
426
+ }
427
+ results[origIdx] = peerResults[i];
428
+ }
429
+ } finally {
430
+ disposeRpcResource(rpcResp);
431
+ }
432
+ return;
433
+ } catch (err) {
434
+ const schedule = isTransientDoReset(err) ? PEER_RETRY_BACKOFF_MS
435
+ : isDoOverloaded(err) ? PEER_OVERLOAD_BACKOFF_MS
436
+ : null;
437
+ if (schedule && attempt < PEER_TRANSIENT_RESET_RETRIES) {
438
+ const backoff = schedule[Math.min(attempt, schedule.length - 1)];
439
+ await new Promise((r) => setTimeout(r, backoff));
440
+ continue;
441
+ }
442
+ // Re-throwing bare loses everything only this frame knows: which
443
+ // sibling ran the shard, how wide it was, and how many attempts
444
+ // it already cost. Callers report the message, so a rejection the
445
+ // platform words as `internal error` arrived at the user with no
446
+ // way to tell a one-off peer from a shard that had exhausted its
447
+ // retries. `cause` keeps the original for anything that inspects
448
+ // errors rather than reads them.
449
+ throw new Error(
450
+ `peer shard ${siblingName} (${bucket.length} task${bucket.length === 1 ? '' : 's'}) `
451
+ + `failed after ${attempt + 1} attempt${attempt === 0 ? '' : 's'}: ${describeError(err)}`,
452
+ { cause: err },
453
+ );
454
+ } finally {
455
+ disposeRpcResource(peerStub);
456
+ }
457
+ }
458
+ });
459
+ }
460
+
461
+ // Dispatch peer shards in bounded phases. A single Promise.all across all
462
+ // shards can create too many simultaneous cold sibling DO starts under
463
+ // concurrent installs. Promise-chain phasing limits scheduler pressure
464
+ // without sleeps, timers, or idle gaps between phases.
465
+ //
466
+ // A phase is a hard barrier: no shard in phase N+1 starts until every
467
+ // shard in phase N returns, so `width@ms` per phase is what separates
468
+ // "the shards are slow" from "there are too many barriers".
469
+ for (let i = 0; i < dispatchers.length; i += FANOUT_PHASE_SIZE) {
470
+ const phase = dispatchers.slice(i, i + FANOUT_PHASE_SIZE);
471
+ const phaseStartedAt = Date.now();
472
+ await Promise.all(phase.map((d) => d()));
473
+ this.opts.onDispatchPhase?.(phase.length, Date.now() - phaseStartedAt);
474
+ }
475
+ return results;
476
+ }
477
+ }
478
+
479
+ /**
480
+ * Stable hash → shard. Uses a fresh djb2 over the key (NOT
481
+ * hashSource) and modulos by peerCount.
482
+ *
483
+ * Why not reuse hashSource: hashSource returns a base-36 string,
484
+ * NOT hex — its alphabet is `[0-9a-z]`. parseInt(str, 16) on a
485
+ * base-36 string aborts at the first non-hex char (any of g-z),
486
+ * which produces extremely poor distribution: keys with the same
487
+ * leading-hex-prefix collide regardless of their suffix. (Seen in
488
+ * the wild: `task-0 .. task-7` all collided onto shard 4.)
489
+ *
490
+ * Deterministic: same key + same peerCount → same shard, every run.
491
+ * Tests use this to predict placement.
492
+ */
493
+ export function hashKeyToShard(key: string, peerCount: number): number {
494
+ if (peerCount <= 1) return 0;
495
+ // djb2, returning an unsigned 32-bit integer — full 2^32 range,
496
+ // no string-format conversion gotchas. peerCount <= MAX_PEER_FANOUT
497
+ // (32) << 2^32, so the modulo distributes uniformly for any input.
498
+ let h = 5381;
499
+ for (let i = 0; i < key.length; i++) {
500
+ h = ((h << 5) + h + key.charCodeAt(i)) | 0;
501
+ }
502
+ return (h >>> 0) % peerCount;
503
+ }
package/src/index.ts ADDED
@@ -0,0 +1,26 @@
1
+ /**
2
+ * @nimbus-sh/fabric — the Cloudflare-specific DO/facet machinery of Nimbus.
3
+ *
4
+ * Root export for workerd contexts. `bindings.ts` imports
5
+ * `cloudflare:workers`, so importing this root module outside workerd fails
6
+ * at resolution; non-workerd consumers (tests, tooling) import the subpath
7
+ * modules they need instead.
8
+ */
9
+
10
+ export * from './alarms.js';
11
+ export * from './bindings.js';
12
+ export * from './ctx-exports.js';
13
+ export * from './facet-image-store.js';
14
+ export * from './fanout-pool.js';
15
+ export * from './inner-do-registry.js';
16
+ export * from './launch-journal.js';
17
+ export * from './launch-pacer.js';
18
+ export * from './loader-ledger.js';
19
+ export * from './loader-pool.js';
20
+ export * from './process-fabric.js';
21
+ export * from './process-host.js';
22
+ export * from './workerd-facet-host.js';
23
+ export * from './ws-hibernation-config.js';
24
+ export * from './vendor/errors.js';
25
+ export * from './vendor/serialize.js';
26
+ export * from './vendor/types.js';
@@ -0,0 +1,58 @@
1
+ /**
2
+ * inner-do-registry.ts — Module-level registry of inner DO classes.
3
+ *
4
+ * `nimbus-wrangler` populates this on each successful buildAndLoad()
5
+ * with the DurableObject classes extracted from the freshly-loaded
6
+ * inner Worker via `worker.getDurableObjectClass(name)`. The supervisor
7
+ * DO (`NimbusSession`) reads it on each inner-DO fetch to synthesize a
8
+ * NimbusDurableObjectNamespace stub for `env.MY_DO`.
9
+ *
10
+ * Why a separate leaf module:
11
+ * Before this extraction, the registry lived inside nimbus-session.ts,
12
+ * which forced nimbus-wrangler.ts to import from nimbus-session.ts —
13
+ * producing the cycle
14
+ * index.ts -> nimbus-session.ts -> nimbus-wrangler.ts -> nimbus-session.ts
15
+ * Promoting the registry to its own leaf breaks that cycle without
16
+ * changing any semantics. The Map identity is preserved across the
17
+ * isolate (it's still process-scoped — module-level state — survives
18
+ * across DO instances in the same workerd process).
19
+ *
20
+ * Key shape:
21
+ * `<supervisor-DO-id>:<binding-name>` — both halves are required to
22
+ * prevent multiple supervisor DOs in the same isolate from clobbering
23
+ * each other's registrations.
24
+ *
25
+ * The values are the inner worker's own Durable Object classes, as
26
+ * `worker.getDurableObjectClass(name)` hands them over: opaque tokens whose
27
+ * only use is `ctx.facets.get(name, { class: cls, id })`, which is exactly what
28
+ * workerd's `DurableObjectClass` models.
29
+ */
30
+
31
+ const _NIMBUS_INNER_DO_CLASSES: Map<string, DurableObjectClass> = new Map();
32
+
33
+ /** Look up a registered inner-DO class. Returns undefined if not found. */
34
+ export function getInnerDoClass(supervisorDoId: string, bindingName: string): DurableObjectClass | undefined {
35
+ return _NIMBUS_INNER_DO_CLASSES.get(supervisorDoId + ':' + bindingName);
36
+ }
37
+
38
+ /**
39
+ * Register an inner DO class for synthesis. Called by nimbus-wrangler.ts
40
+ * after each successful buildAndLoad() with the class extracted from
41
+ * the fresh worker stub. Keys are `<doId>:<bindingName>` so multiple
42
+ * supervisor DOs don't collide.
43
+ */
44
+ export function registerInnerDoClass(
45
+ supervisorDoId: string,
46
+ bindingName: string,
47
+ cls: DurableObjectClass,
48
+ ): void {
49
+ _NIMBUS_INNER_DO_CLASSES.set(supervisorDoId + ':' + bindingName, cls);
50
+ }
51
+
52
+ /** Clear all registrations belonging to a supervisor DO (called on rebuild). */
53
+ export function clearInnerDoClasses(supervisorDoId: string): void {
54
+ const prefix = supervisorDoId + ':';
55
+ for (const k of _NIMBUS_INNER_DO_CLASSES.keys()) {
56
+ if (k.startsWith(prefix)) _NIMBUS_INNER_DO_CLASSES.delete(k);
57
+ }
58
+ }