@nimbus-sh/fabric 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/README.md +19 -1
  2. package/dist/bindings.d.ts +11 -0
  3. package/dist/bindings.d.ts.map +1 -1
  4. package/dist/bindings.js +19 -8
  5. package/dist/connections.d.ts +1 -1
  6. package/dist/connections.js +1 -1
  7. package/dist/do-calls.d.ts +71 -9
  8. package/dist/do-calls.d.ts.map +1 -1
  9. package/dist/do-calls.js +148 -22
  10. package/dist/facet-pool.d.ts +3 -0
  11. package/dist/facet-pool.d.ts.map +1 -1
  12. package/dist/facet-pool.js +3 -0
  13. package/dist/fanout.d.ts.map +1 -1
  14. package/dist/fanout.js +6 -3
  15. package/dist/fenced-work.d.ts +3 -3
  16. package/dist/host-dispatch.d.ts +12 -4
  17. package/dist/host-dispatch.d.ts.map +1 -1
  18. package/dist/host-dispatch.js +11 -4
  19. package/dist/image-store.js +2 -2
  20. package/dist/inner-do-registry.d.ts +9 -0
  21. package/dist/inner-do-registry.d.ts.map +1 -1
  22. package/dist/inner-do-registry.js +35 -0
  23. package/dist/isolate-pool.d.ts +8 -1
  24. package/dist/isolate-pool.d.ts.map +1 -1
  25. package/dist/isolate-pool.js +9 -7
  26. package/dist/process-fabric.d.ts +16 -5
  27. package/dist/process-fabric.d.ts.map +1 -1
  28. package/dist/process-fabric.js +18 -6
  29. package/dist/process-host.d.ts +5 -0
  30. package/dist/process-host.d.ts.map +1 -1
  31. package/dist/process-host.js +13 -11
  32. package/dist/supervisor-props.d.ts +47 -0
  33. package/dist/supervisor-props.d.ts.map +1 -0
  34. package/dist/supervisor-props.js +37 -0
  35. package/dist/workerd-facet-host.d.ts +4 -17
  36. package/dist/workerd-facet-host.d.ts.map +1 -1
  37. package/dist/workerd-facet-host.js +98 -32
  38. package/package.json +6 -6
  39. package/src/bindings.ts +25 -10
  40. package/src/connections.ts +1 -1
  41. package/src/do-calls.ts +182 -25
  42. package/src/facet-pool.ts +5 -0
  43. package/src/fanout.ts +6 -3
  44. package/src/fenced-work.ts +3 -3
  45. package/src/host-dispatch.ts +22 -7
  46. package/src/image-store.ts +2 -2
  47. package/src/inner-do-registry.ts +31 -0
  48. package/src/isolate-pool.ts +16 -8
  49. package/src/process-fabric.ts +33 -10
  50. package/src/process-host.ts +18 -11
  51. package/src/supervisor-props.ts +56 -0
  52. package/src/workerd-facet-host.ts +99 -37
@@ -13,9 +13,11 @@
13
13
  * `HostedProcess` and never imports this file.
14
14
  */
15
15
  import { disposeRpcResource } from '@nimbus-sh/platform/rpc-dispose.js';
16
+ import { StorageLedger, forgetFacetStorage } from '@nimbus-sh/core/runtime/storage-ledger.js';
16
17
  import { getCtxExports, stagedBootAssembler, supervisorEntrypoint, supervisorEntrypointName, } from './composition.js';
17
18
  import { assertModuleMapWithinCodeLimit, beginLoaderFetch, facetNameCount, facetNameCountDurable, recordFacetNameMinted, recordLoaderId, withDynamicWorkerCapNamed, withFacetBudgetNamed, } from './budgets.js';
18
19
  import { RESIDENT_PROCESS_CLASS, residentLoaderConfig, } from './process-fabric.js';
20
+ import { supervisorLoaderKey } from './supervisor-props.js';
19
21
  export function getNimbusCtxExports() {
20
22
  const ctxExports = getCtxExports();
21
23
  if (!ctxExports || typeof ctxExports !== 'object') {
@@ -35,7 +37,9 @@ export async function createLoadedWorkerEntrypoint(ctxExports, supervisor, stage
35
37
  }
36
38
  return await ctxExports.NimbusLoadedEntrypoint({
37
39
  props: {
38
- key: `nimbus-process:${supervisor.doId}:${supervisor.pid}`,
40
+ // The entrypoint's loader outlives this instance, and a warm worker keeps
41
+ // the SUPERVISOR binding it was built with.
42
+ key: supervisorLoaderKey(`nimbus-process:${supervisor.doId}:${supervisor.pid}`, supervisor),
39
43
  name,
40
44
  depth: 0,
41
45
  supervisor,
@@ -126,37 +130,38 @@ export const DURABLE_FACET_NAME_PREFIX = 'app-slot-';
126
130
  * Slot books, per hosting actor, because the facet index is per Durable
127
131
  * Object.
128
132
  *
129
- * Keyed weakly off `ctx`, and that is sound rather than lossy: a facet cannot
130
- * outlive the Durable Object hosting it, so a book that goes away with its
131
- * host describes nothing that still exists. A fresh incarnation restarts at
132
- * slot 0 and re-attaches to the SQLite a previous incarnation left there —
133
- * which is safe for the reason the store is sealed until it has reconciled.
134
- * Its persisted cursor is either datable against the current authority, in
135
- * which case the ACQUIRE delta brings it current, or it carries a different
136
- * VFS epoch, in which case `invalidatedSince` can only answer poison and the
137
- * whole store is dropped. A process therefore cannot boot onto a previous
138
- * tenant's filesystem even when release never ran.
133
+ * Keyed weakly off `ctx`, so a book describes one incarnation. A facet that
134
+ * is still running when that incarnation ends outlives it (measured: a timer
135
+ * or an outgoing call keeps it going), and a `get` of its name with a new
136
+ * class then resets the whole object. So a fresh incarnation's first get of
137
+ * each name ends whatever runs there first: a minted `proc-slot-` name is
138
+ * deleted, which also wipes the storage a previous incarnation left, and an
139
+ * explicit name is aborted, which keeps it.
139
140
  *
140
- * The book names only the `proc-slot-` space. Durable `app-slot-` names are
141
- * allocated against DO storage instead (their owner survives a reset), so a
142
- * fresh incarnation's `next` starting at 0 can never collide with them even
141
+ * The book allocates only the `proc-slot-` space. Durable `app-slot-` names
142
+ * are allocated against DO storage instead (their owner survives a reset), so
143
+ * a fresh incarnation's `next` starting at 0 can never collide with them even
143
144
  * before the durable ledger is adopted.
144
145
  */
145
146
  const slotBooks = new WeakMap();
146
147
  function slotBook(ctx) {
147
148
  let book = slotBooks.get(ctx);
148
149
  if (!book) {
149
- book = { free: [], next: 0, held: new Map() };
150
+ book = { free: [], next: 0, held: new Map(), live: new Set() };
150
151
  slotBooks.set(ctx, book);
151
152
  }
152
153
  return book;
153
154
  }
154
- /** Take a slot for `pid`, reusing a returned one before minting a new name. */
155
+ /**
156
+ * Take a slot for `pid`, reusing a returned one before minting a new name.
157
+ * `minted` names may still hold storage a previous incarnation of this actor
158
+ * left there, so the caller deletes it before the first get.
159
+ */
155
160
  function acquireSlot(ctx, pid) {
156
161
  const book = slotBook(ctx);
157
162
  const existing = book.held.get(pid);
158
163
  if (existing !== undefined)
159
- return existing;
164
+ return { slot: existing, minted: false };
160
165
  const reused = book.free.length > 0;
161
166
  const slot = reused ? book.free.shift() : book.next++;
162
167
  book.held.set(pid, slot);
@@ -164,7 +169,7 @@ function acquireSlot(ctx, pid) {
164
169
  // in the budgets ledger (see budgets.ts).
165
170
  if (!reused)
166
171
  recordFacetNameMinted(ctx, book.next);
167
- return slot;
172
+ return { slot, minted: !reused };
168
173
  }
169
174
  /** Return `pid`'s slot to the free list. */
170
175
  function releaseSlot(ctx, pid) {
@@ -177,14 +182,34 @@ function releaseSlot(ctx, pid) {
177
182
  book.free.sort((a, b) => a - b);
178
183
  }
179
184
  /**
180
- * Drop one facet's SQLite by name — the ONLY call site that may delete facet
181
- * storage. `spawnResident` releases ephemeral processes with abort+delete
182
- * (storage is slot-reuse hygiene) and durable ones with abort alone (the
183
- * storage IS the durable application's state); explicit removal arrives here
184
- * through the coordinator's durable-slot book, owner-checked.
185
+ * Drop one facet's SQLite by name, and its row in the session's storage
186
+ * ledger (N18) in the same step: the only way a facet database is deleted.
187
+ * `spawnResident` releases ephemeral processes with abort+delete (storage is
188
+ * slot-reuse hygiene) and durable ones with abort alone (the storage IS the
189
+ * durable application's state); explicit removal arrives here through the
190
+ * coordinator's durable-slot book, owner-checked.
185
191
  */
192
+ /** The session's storage ledger (N18), over this actor's SQL; null where it has none. */
193
+ function sessionLedger(ctx) {
194
+ const sql = ctx.storage?.sql;
195
+ return sql ? new StorageLedger(sql) : null;
196
+ }
197
+ const facetNames = new WeakMap();
198
+ function facetOfPid(ctx) {
199
+ let names = facetNames.get(ctx);
200
+ if (!names)
201
+ facetNames.set(ctx, names = new Map());
202
+ return names;
203
+ }
204
+ /** The facet a running resident process `pid` lives in on this actor, for its storage ledger row. */
205
+ export function residentFacetOf(ctx, pid) {
206
+ return facetNames.get(ctx)?.get(pid);
207
+ }
186
208
  export function deleteFacetStorage(ctx, name) {
187
209
  facetContainer(ctx).delete(name);
210
+ const sql = ctx.storage?.sql;
211
+ if (sql)
212
+ forgetFacetStorage(sql, name);
188
213
  }
189
214
  /**
190
215
  * The process surface of one hosting actor: how a resident process comes
@@ -241,8 +266,15 @@ function spawnResident(ctx, env, disk, supervisor, params) {
241
266
  throw new Error(`Nimbus: an explicit facet name must carry the '${DURABLE_FACET_NAME_PREFIX}' `
242
267
  + `prefix, got '${explicit.name}'`);
243
268
  }
244
- const slot = explicit ? undefined : acquireSlot(ctx, params.pid);
269
+ const grant = explicit ? undefined : acquireSlot(ctx, params.pid);
270
+ const slot = grant?.slot;
245
271
  const name = explicit ? explicit.name : residentFacetName(slot);
272
+ if (grant?.minted) {
273
+ try {
274
+ deleteFacetStorage(ctx, name);
275
+ }
276
+ catch { /* nothing stored under this name */ }
277
+ }
246
278
  // The start callback is the ONLY way this facet is ever created, and it
247
279
  // fires AT MOST ONCE. Every later use goes through the stub below, so the
248
280
  // callback running a second time means the facet was released or died —
@@ -262,8 +294,18 @@ function spawnResident(ctx, env, disk, supervisor, params) {
262
294
  evaluated = true;
263
295
  return { class: residentProcessClass(ctx, env, disk, supervisor, params) };
264
296
  };
297
+ const book = slotBook(ctx);
298
+ const ledger = sessionLedger(ctx);
265
299
  let facet;
266
300
  try {
301
+ // N18: the fill is admitted, and recorded under the facet's name, before
302
+ // the facet exists; a refusal leaves no facet.
303
+ if (ledger !== null && params.storageBytes !== undefined)
304
+ ledger.fill(name, params.storageBytes);
305
+ // get() with a new class on a facet an earlier incarnation left running resets this object.
306
+ if (explicit && !book.live.has(name)) {
307
+ facets.abort(name, new Error('Nimbus: a new incarnation takes this facet name'));
308
+ }
267
309
  facet = facets.get(name, start);
268
310
  }
269
311
  catch (error) {
@@ -271,16 +313,22 @@ function spawnResident(ctx, env, disk, supervisor, params) {
271
313
  releaseSlot(ctx, params.pid);
272
314
  throw withFacetBudgetNamed(facetNameCount(ctx), error);
273
315
  }
316
+ if (explicit)
317
+ book.live.add(name);
318
+ facetOfPid(ctx).set(params.pid, name);
274
319
  let disposed = false;
275
320
  const release = async () => {
276
321
  if (disposed)
277
322
  return;
278
323
  disposed = true;
279
324
  released = true;
325
+ facetOfPid(ctx).delete(params.pid);
280
326
  try {
281
327
  facets.abort(name, new Error('Nimbus: resident process released'));
282
328
  }
283
329
  catch { /* already gone */ }
330
+ if (explicit)
331
+ book.live.delete(name);
284
332
  // The two release classes: an ephemeral facet's SQLite is slot-reuse
285
333
  // hygiene — the name is handed out again, so the store must not be — and
286
334
  // a durable one's is the application itself: abort ends the process, the
@@ -288,7 +336,7 @@ function spawnResident(ctx, env, disk, supervisor, params) {
288
336
  // deleteFacetStorage call ever drops it.
289
337
  if (!explicit?.durable) {
290
338
  try {
291
- facets.delete(name);
339
+ deleteFacetStorage(ctx, name);
292
340
  }
293
341
  catch { /* already gone */ }
294
342
  }
@@ -299,7 +347,11 @@ function spawnResident(ctx, env, disk, supervisor, params) {
299
347
  };
300
348
  let started;
301
349
  try {
302
- started = facet.startProcess(params.startArgs);
350
+ // The allowance the ledger admitted, for the facet's store to keep under.
351
+ const startArgs = ledger !== null && params.storageBytes !== undefined && params.startArgs !== null && typeof params.startArgs === 'object'
352
+ ? { ...params.startArgs, storage: { facet: name, grant: params.storageBytes } }
353
+ : params.startArgs;
354
+ started = facet.startProcess(startArgs);
303
355
  }
304
356
  catch (error) {
305
357
  void release();
@@ -309,7 +361,18 @@ function spawnResident(ctx, env, disk, supervisor, params) {
309
361
  // this one, and it is annotated AFTER awaiting the ledger — the first
310
362
  // failure of a fresh incarnation must compare against the persisted count,
311
363
  // not the zero its adoption read has not yet replaced.
312
- started = started.catch(async (error) => {
364
+ started = started.then((payload) => {
365
+ // Once the facet is up (N18) its row is the cap its store keeps under
366
+ // (what it measures plus what it may still grow into), or what it
367
+ // measures if that is more (overshoot).
368
+ const { databaseSize: size, storageCap: cap } = (payload ?? {});
369
+ const measured = typeof size === 'number' && Number.isFinite(size) ? size : null;
370
+ const capped = typeof cap === 'number' && Number.isFinite(cap) ? cap : null;
371
+ const row = capped !== null ? Math.max(capped, measured ?? 0) : measured;
372
+ if (ledger !== null && row !== null)
373
+ ledger.reportSize(name, row);
374
+ return payload;
375
+ }, async (error) => {
313
376
  throw withFacetBudgetNamed(await facetNameCountDurable(ctx), error);
314
377
  });
315
378
  // A caller reads whichever of `started` and the lifecycle it needs, so keep
@@ -339,11 +402,14 @@ function residentProcessClass(ctx, env, disk, supervisor, params) {
339
402
  throw new Error('Nimbus: env.LOADER binding missing or invalid. Resident processes require '
340
403
  + 'the Worker Loader binding; add it via worker_loaders in wrangler.jsonc.');
341
404
  }
405
+ // A warm worker keeps the SUPERVISOR binding it was built with, and the
406
+ // loader outlives this instance.
407
+ const loaderKey = supervisorLoaderKey(params.workerKey, supervisor);
342
408
  try {
343
409
  const worker = loader
344
- .get(params.workerKey, () => residentWorkerConfig(env, disk, supervisor, params.boot))
410
+ .get(loaderKey, () => residentWorkerConfig(env, disk, supervisor, params.boot))
345
411
  .getDurableObjectClass(RESIDENT_PROCESS_CLASS);
346
- recordLoaderId(ctx, params.workerKey);
412
+ recordLoaderId(ctx, loaderKey);
347
413
  return worker;
348
414
  }
349
415
  catch (error) {
@@ -356,7 +422,7 @@ async function runOneShot(ctx, env, supervisor, params, consume) {
356
422
  throw new Error('Nimbus: env.LOADER binding missing or invalid. Running a program requires '
357
423
  + 'the Worker Loader binding; add it via worker_loaders in wrangler.jsonc.');
358
424
  }
359
- const supervisorRpc = supervisorEntrypoint();
425
+ const supervisorRpc = supervisorEntrypoint(undefined, supervisor.route?.supervisorEntrypoint);
360
426
  let supervisorBinding;
361
427
  let worker;
362
428
  let entrypoint;
@@ -439,9 +505,9 @@ export async function residentWorkerConfig(env, disk, supervisor, boot) {
439
505
  ? await stagedBootAssembler()(env, boot.stage)
440
506
  : await residentLoaderConfig(boot.code, disk());
441
507
  assertModuleMapWithinCodeLimit(configModules(config));
442
- const supervisorRpc = supervisorEntrypoint();
508
+ const supervisorRpc = supervisorEntrypoint(undefined, supervisor.route?.supervisorEntrypoint);
443
509
  if (!supervisorRpc) {
444
- throw new Error(`Nimbus: ctx.exports.${supervisorEntrypointName() ?? '<supervisor entrypoint>'} unavailable`);
510
+ throw new Error(`Nimbus: ctx.exports.${supervisor.route?.supervisorEntrypoint ?? supervisorEntrypointName() ?? '<supervisor entrypoint>'} unavailable`);
445
511
  }
446
512
  return { ...config, env: { SUPERVISOR: supervisorRpc({ props: supervisor }) } };
447
513
  }
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@nimbus-sh/fabric",
3
- "version": "0.7.0",
4
- "description": "The Cloudflare half of Nimbus — Durable Object facet hosting, dynamic-worker loader pools, the resident-process fabric, and DO alarm/hibernation machinery.",
3
+ "version": "0.8.0",
4
+ "description": "The Cloudflare half of Nimbus \u2014 Durable Object facet hosting, dynamic-worker loader pools, the resident-process fabric, and DO alarm/hibernation machinery.",
5
5
  "keywords": [
6
6
  "cloudflare",
7
7
  "workers",
@@ -45,13 +45,13 @@
45
45
  "LICENSE"
46
46
  ],
47
47
  "scripts": {
48
- "build": "tsc -p tsconfig.json --noCheck --noEmit false --declaration true --declarationMap true --outDir dist --rootDir src",
49
- "prepack": "bun run build",
48
+ "build": "node ../../scripts/clean-dist.mjs && tsc -p tsconfig.json --noCheck --noEmit false --declaration true --declarationMap true --outDir dist --rootDir src",
49
+ "prepublishOnly": "bun ../../scripts/dist-integrity.mjs --publish",
50
50
  "typecheck": "tsc --noEmit"
51
51
  },
52
52
  "dependencies": {
53
- "@nimbus-sh/core": "^0.11.0",
54
- "@nimbus-sh/platform": "^0.5.0",
53
+ "@nimbus-sh/core": "^0.13.0",
54
+ "@nimbus-sh/platform": "^0.6.0",
55
55
  "zod": "^4.4.3"
56
56
  },
57
57
  "devDependencies": {
package/src/bindings.ts CHANGED
@@ -26,8 +26,8 @@
26
26
  import { WorkerEntrypoint } from 'cloudflare:workers';
27
27
  import { z } from 'zod/v4';
28
28
  import { disposeRpcResource, useRpcResource } from '@nimbus-sh/platform/rpc-dispose.js';
29
- import { hostNamespace, supervisorEntrypoint, supervisorEntrypointName } from './composition.js';
30
- import { stagedBootAssembler } from './composition.js';
29
+ import { hostNamespace, supervisorEntrypoint, supervisorEntrypointName, stagedBootAssembler } from './composition.js';
30
+ import type { HostRoute } from './composition.js';
31
31
  import { hostNamespaceBinding, hostOpDispatch, type HostOpDispatch } from './host-dispatch.js';
32
32
  import { assertModuleMapWithinCodeLimit } from './budgets.js';
33
33
  import type { EntrypointLoopbackFactory } from './composition.js';
@@ -114,6 +114,8 @@ interface NimbusAssetsProps {
114
114
  assetsDir?: string;
115
115
  /** Supervisor DO id whose VFS holds the assets. */
116
116
  doId?: string;
117
+ /** The way back to that DO, minted with the binding. */
118
+ route?: HostRoute;
117
119
  }
118
120
 
119
121
  /**
@@ -164,9 +166,9 @@ export class NimbusAssetsRPC extends WorkerEntrypoint<object, NimbusAssetsProps>
164
166
  let stub: DurableObjectStub | null = null;
165
167
  let dispatch: HostOpDispatch;
166
168
  try {
167
- const ns = hostNamespaceBinding(this.env, 'NimbusAssetsRPC');
169
+ const ns = hostNamespaceBinding(this.env, 'NimbusAssetsRPC', props.route);
168
170
  stub = ns.get(ns.idFromString(doId));
169
- dispatch = hostOpDispatch(stub, 'NimbusAssetsRPC');
171
+ dispatch = hostOpDispatch(stub, 'NimbusAssetsRPC', props.route);
170
172
  } catch (e) {
171
173
  disposeRpcResource(stub);
172
174
  return new Response(`ASSETS binding not wired: ${e instanceof Error ? e.message : String(e)}`, { status: 500 });
@@ -307,6 +309,12 @@ const _NIMBUS_LOADED_CODES: Map<string, WorkerCode> = new Map();
307
309
  const _LOADED_CODES_MAX = 32;
308
310
  let _loadedCodesEvictions = 0;
309
311
 
312
+ const HostRouteSchema = z.object({
313
+ supervisorEntrypoint: z.string().min(1),
314
+ hostNamespace: z.string().min(1),
315
+ hostDispatchMethod: z.string().min(1),
316
+ });
317
+
310
318
  const NimbusLoadedEntrypointPropsSchema = z.object({
311
319
  key: z.string().min(1),
312
320
  name: z.string().nullable().optional(),
@@ -315,6 +323,9 @@ const NimbusLoadedEntrypointPropsSchema = z.object({
315
323
  doId: z.string().min(1),
316
324
  pid: z.number().int().nonnegative(),
317
325
  writerId: z.string().uuid(),
326
+ route: HostRouteSchema.optional(),
327
+ /** The host instance's delivery incarnation (ResidentSupervisorProps). */
328
+ hostIncarnation: z.string().uuid().optional(),
318
329
  }).optional(),
319
330
  /**
320
331
  * Staged-artifact spec, for a ONE-SHOT run. The module map — ~23 MB for
@@ -532,11 +543,12 @@ export class NimbusLoadedEntrypoint extends WorkerEntrypoint<NimbusLoaderShimEnv
532
543
 
533
544
  async _supervisorBinding(props: NimbusLoadedEntrypointProps): Promise<unknown> {
534
545
  if (!props.supervisor) return undefined;
535
- const factory = supervisorEntrypoint(ctxExportsOf(this.ctx));
546
+ // The route names the entrypoint too: this stateless hop may run in an
547
+ // isolate whose composition is not the host's.
548
+ const name = props.supervisor.route?.supervisorEntrypoint ?? supervisorEntrypointName();
549
+ const factory = supervisorEntrypoint(ctxExportsOf(this.ctx), name ?? undefined);
536
550
  if (!factory) {
537
- throw new Error(
538
- `Nimbus: ctx.exports.${supervisorEntrypointName() ?? '<supervisor entrypoint>'} unavailable`,
539
- );
551
+ throw new Error(`Nimbus: ctx.exports.${name ?? '<supervisor entrypoint>'} unavailable`);
540
552
  }
541
553
  return await factory({ props: props.supervisor });
542
554
  }
@@ -691,6 +703,8 @@ export class NimbusLoadedEntrypoint extends WorkerEntrypoint<NimbusLoaderShimEnv
691
703
  interface NimbusDoNamespaceProps {
692
704
  bindingName?: string;
693
705
  supervisorDoId?: string;
706
+ /** The way back to the supervisor DO, minted with the binding. */
707
+ route?: HostRoute;
694
708
  }
695
709
 
696
710
  /**
@@ -755,6 +769,7 @@ export class NimbusDurableObjectNamespace extends WorkerEntrypoint<unknown, Nimb
755
769
  props: {
756
770
  bindingName: props.bindingName,
757
771
  supervisorDoId: props.supervisorDoId,
772
+ route: props.route,
758
773
  id: String(id),
759
774
  },
760
775
  });
@@ -789,9 +804,9 @@ export class NimbusDOStub extends WorkerEntrypoint<object, NimbusDoStubProps> {
789
804
  let stub: DurableObjectStub | null = null;
790
805
  let dispatch: HostOpDispatch;
791
806
  try {
792
- const ns = hostNamespaceBinding(this.env ?? {}, 'NimbusDOStub');
807
+ const ns = hostNamespaceBinding(this.env ?? {}, 'NimbusDOStub', props.route);
793
808
  stub = ns.get(ns.idFromString(supervisorDoId));
794
- dispatch = hostOpDispatch(stub, 'NimbusDOStub');
809
+ dispatch = hostOpDispatch(stub, 'NimbusDOStub', props.route);
795
810
  } catch (e) {
796
811
  disposeRpcResource(stub);
797
812
  return new Response(`Nimbus: ${e instanceof Error ? e.message : String(e)}`, { status: 500 });
@@ -3,7 +3,7 @@
3
3
  * state over the WebSocket attachment.
4
4
  *
5
5
  * Specified from Proteus's DeviceSocketHub (`cf-backend/src/user/device-hub.ts`)
6
- * and CLI rpc gate (`cf-backend/src/cli/rpc-gate.ts`), which split the
6
+ * and its original CLI rpc gate (`cf-backend/src/cli/rpc-gate.ts`), which split the
7
7
  * pattern into its two halves:
8
8
  * - a TAG is the immutable-at-accept lookup key and authorization — it
9
9
  * rides the hibernation state, which is why the rpc gate persists auth
package/src/do-calls.ts CHANGED
@@ -3,12 +3,14 @@
3
3
  * the one property that decides whether a retry is safe.
4
4
  *
5
5
  * Both consumers asked for this. Proteus hand-wrote the retry
6
- * (`cf-backend/src/lib/do-rpc.ts`) with the rule its header states:
6
+ * (originally `cf-backend/src/lib/do-rpc.ts`) with the rule its header states:
7
7
  * "An operation that appends, sends, charges or mints is never wrapped: a
8
8
  * dropped call there may already have run, so a retry is a correctness bug
9
9
  * wearing resilience as a costume." agent-core has no retry machinery at all
10
10
  * and its backlog calls the gap "the most production-proven gap in the
11
11
  * corpus". Here the rule is a type: `idempotent` retries, `mutating` cannot.
12
+ * A mutation earns a retry only by carrying an identity its callee applies
13
+ * at most once; it is then `idempotent` by construction (see `mutating`).
12
14
  *
13
15
  * What the platform contract requires, and this keeps:
14
16
  * - a FRESH stub per attempt. Cloudflare documents that many exceptions
@@ -20,6 +22,11 @@
20
22
  * object is what overloaded it.
21
23
  * - attempts and backoff are the consumer-proven bounds: 3 attempts total,
22
24
  * full-jitter delays in [0, 2**attempt * 60ms).
25
+ * - an `idempotent` call may also be HEDGED (`hedgeAfterMs`): an attempt
26
+ * that has not answered by then is joined by the same call on a fresh
27
+ * stub, both left running, the first success taken. Hedges count
28
+ * against the attempts, and a callee that joins a repeat to the call it
29
+ * is already serving makes one that did arrive cost nothing.
23
30
  *
24
31
  * The resolver MINTS a stub per call and the verb disposes each one it
25
32
  * minted — that ownership is what makes the fresh-stub retry real.
@@ -47,10 +54,38 @@ export interface DoCallRetryPolicy {
47
54
  maxAttempts?: number;
48
55
  baseDelayMs?: number;
49
56
  /**
50
- * Called once per retry, before its backoff delay, with the failure the
51
- * retry is answering. The consumer's logging seam: Proteus's hand-rolled
52
- * predecessor logged every retry so a flaky object is visible in Workers
53
- * Logs rather than silently absorbed, and `operation` names it there.
57
+ * No repeat — retry or hedge — starts once this long has passed since the
58
+ * first attempt did; the failure in hand surfaces instead. A mutation made
59
+ * repeatable by an identity its callee dedupes needs it: the callee keeps
60
+ * what answers a repeat for a bounded time, so the caller's repeats must
61
+ * stop well inside it. Unbounded when absent.
62
+ */
63
+ retryWindowMs?: number;
64
+ /**
65
+ * Hedge an attempt that has not answered after this long: send the same
66
+ * call again on a fresh stub while the first stays in flight. The caller
67
+ * gets the first success; an answer after it is disposed and dropped. That
68
+ * is only harmless when a second delivery of the call changes nothing — a
69
+ * read — and cheap only when the callee joins a repeat to the call it is
70
+ * already serving. A hedge is an attempt: it counts against `maxAttempts`,
71
+ * and has its own deadline. Never hedged when absent.
72
+ *
73
+ * With attempts overlapping, a failure decides less. A retryable one is
74
+ * retried after its backoff while attempts remain, and `overloaded` stops
75
+ * every further repeat; either way the call keeps waiting on the attempts
76
+ * still in flight, and fails with the last failure only once none is. Any
77
+ * other failure is the callee's answer — the call ran — and is the call's
78
+ * at once.
79
+ */
80
+ hedgeAfterMs?: number;
81
+ /**
82
+ * Called once per retry, as it starts — after its backoff delay — with the
83
+ * failure the retry is answering. A retry a hedge made unnecessary, or
84
+ * that no attempt is left for, is never announced. The consumer's logging
85
+ * seam: Proteus's hand-rolled predecessor logged every retry so a flaky
86
+ * object is visible in Workers Logs rather than silently absorbed, and
87
+ * `operation` names it there. A callback that throws fails the call with
88
+ * its error.
54
89
  */
55
90
  onRetry?(info: DoCallRetryInfo): void;
56
91
  }
@@ -60,7 +95,10 @@ export interface DoCallRetryPolicy {
60
95
  export interface DoCallRetryInfo {
61
96
  operation: string;
62
97
  classification: DoCallClass;
63
- /** The 1-based attempt that failed; the retry about to run is attempt+1. */
98
+ /**
99
+ * The 1-based number of the attempt that failed. Without hedging the retry
100
+ * is attempt+1; with it, other attempts may have started in between.
101
+ */
64
102
  attempt: number;
65
103
  maxAttempts: number;
66
104
  error: unknown;
@@ -102,11 +140,20 @@ export class DoCallError extends Error {
102
140
 
103
141
  /**
104
142
  * Call another Durable Object with an operation that is safe to repeat: a
105
- * read, or a converge-to-a-value write. Transient failures retry on a fresh
106
- * stub with full-jitter backoff; overloaded and permanent failures surface
107
- * unchanged, as does the last error at exhaustion.
143
+ * read, a converge-to-a-value write, or a mutation carrying an identity its
144
+ * callee applies at most once (see {@link mutating}). Transient failures
145
+ * retry on a fresh stub with full-jitter backoff; overloaded and permanent
146
+ * failures surface unchanged, as does the last error once no attempt may be
147
+ * repeated — attempts spent, or the policy's retry window closed. With
148
+ * `hedgeAfterMs`, an attempt still unanswered by then is joined by another
149
+ * on a fresh stub, and the first answer is taken: a success, or the
150
+ * callee's own error. A transient or overloaded failure then ends the call
151
+ * only once no attempt is left in flight.
152
+ *
153
+ * A failure of the resolver or of `onRetry` is the caller's own, and fails
154
+ * the call with it at once.
108
155
  */
109
- export async function idempotent<S, T>(
156
+ export function idempotent<S, T>(
110
157
  operation: string,
111
158
  stub: DoStubResolver<S>,
112
159
  call: (stub: S) => Promise<T>,
@@ -114,23 +161,118 @@ export async function idempotent<S, T>(
114
161
  ): Promise<T> {
115
162
  const maxAttempts = policy.maxAttempts ?? MAX_ATTEMPTS;
116
163
  const baseDelayMs = policy.baseDelayMs ?? BASE_DELAY_MS;
117
- for (let attempt = 1; ; attempt++) {
118
- const minted = await stub();
119
- try {
120
- const result = await call(minted);
121
- disposeRpcResource(minted);
122
- return result;
123
- } catch (error) {
124
- // A stub that threw may be permanently broken; it is never reused.
125
- disposeRpcResource(minted);
164
+ const { hedgeAfterMs, retryWindowMs } = policy;
165
+ const startedAt = Date.now();
166
+ // The executor form: fabric's library target predates Promise.withResolvers.
167
+ return new Promise<T>((resolve, reject) => {
168
+ const hedges = new Set<ReturnType<typeof setTimeout>>();
169
+ let started = 0;
170
+ // Attempts in flight, or backing off before their retry: while one is,
171
+ // a failure is not the call's answer.
172
+ let live = 0;
173
+ // An attempt was shed as overloaded: nothing is repeated after it.
174
+ let refused = false;
175
+ let settled = false;
176
+
177
+ const settle = (answer: () => void): void => {
178
+ if (settled) return;
179
+ settled = true;
180
+ for (const timer of hedges) clearTimeout(timer);
181
+ hedges.clear();
182
+ answer();
183
+ };
184
+ /** May another attempt start at `at`? */
185
+ const canRepeat = (at: number): boolean =>
186
+ !settled && !refused && started < maxAttempts
187
+ && (retryWindowMs === undefined || at - startedAt <= retryWindowMs);
188
+ /** `error` ended an attempt that will not be repeated: the call's answer, once nothing else is live. */
189
+ const exhausted = <E>(error: E): void => {
190
+ if (live === 0) settle(() => reject(error));
191
+ };
192
+
193
+ /** A failed attempt, numbered: repeat it after its backoff, or let it stand. */
194
+ const failed = async <E>(number: number, error: E): Promise<void> => {
195
+ if (settled) return;
126
196
  const classification = classifyDoCall(error);
127
- if (!isRetryableDoCall(classification) || attempt >= maxAttempts) throw error;
128
- policy.onRetry?.({ operation, classification, attempt, maxAttempts, error });
129
- await new Promise<void>((resolve) => {
130
- setTimeout(resolve, Math.floor(Math.random() * 2 ** attempt * baseDelayMs));
197
+ if (classification === 'overloaded') {
198
+ // A shed call is no answer: nothing more is sent, and an attempt
199
+ // still in flight may yet answer.
200
+ refused = true;
201
+ exhausted(error);
202
+ return;
203
+ }
204
+ if (!isRetryableDoCall(classification)) {
205
+ // The call ran and its answer is this error — ENOENT is a read's answer as much as bytes are.
206
+ settle(() => reject(error));
207
+ return;
208
+ }
209
+ const delayMs = Math.floor(Math.random() * 2 ** number * baseDelayMs);
210
+ if (!canRepeat(Date.now() + delayMs)) {
211
+ exhausted(error);
212
+ return;
213
+ }
214
+ live++;
215
+ await new Promise<void>((wake) => {
216
+ setTimeout(wake, delayMs);
131
217
  });
132
- }
133
- }
218
+ live--;
219
+ // A hedge may have taken the last attempt, or answered, meanwhile.
220
+ if (!canRepeat(Date.now())) {
221
+ exhausted(error);
222
+ return;
223
+ }
224
+ policy.onRetry?.({ operation, classification, attempt: number, maxAttempts, error });
225
+ attempt();
226
+ };
227
+
228
+ const run = async (): Promise<void> => {
229
+ const number = ++started;
230
+ live++;
231
+ const minted = await stub();
232
+ if (settled) {
233
+ live--;
234
+ disposeRpcResource(minted);
235
+ return;
236
+ }
237
+ const hedge = hedgeAfterMs === undefined ? undefined : setTimeout(() => {
238
+ if (hedge !== undefined) hedges.delete(hedge);
239
+ if (canRepeat(Date.now())) attempt();
240
+ }, hedgeAfterMs);
241
+ if (hedge !== undefined) hedges.add(hedge);
242
+ /** This attempt has its answer: it hedges no more, and its stub goes. */
243
+ const answered = (): void => {
244
+ if (hedge !== undefined) {
245
+ clearTimeout(hedge);
246
+ hedges.delete(hedge);
247
+ }
248
+ live--;
249
+ // A stub that threw may be permanently broken; none is ever reused.
250
+ disposeRpcResource(minted);
251
+ };
252
+ let result: T;
253
+ try {
254
+ result = await call(minted);
255
+ } catch (error) {
256
+ answered();
257
+ await failed(number, error);
258
+ return;
259
+ }
260
+ answered();
261
+ if (settled) {
262
+ // Another attempt answered first: this answer is dropped, so nothing of it is kept.
263
+ disposeRpcResource(result);
264
+ return;
265
+ }
266
+ settle(() => resolve(result));
267
+ };
268
+
269
+ /** Start an attempt. Whatever it throws outside the call itself fails the call. */
270
+ const attempt = (): void => {
271
+ run().catch((error) => settle(() => reject(error)));
272
+ };
273
+
274
+ attempt();
275
+ });
134
276
  }
135
277
 
136
278
  /**
@@ -138,6 +280,21 @@ export async function idempotent<S, T>(
138
280
  * or mints. NEVER retried — a dropped call may already have run. Failure
139
281
  * surfaces as a {@link DoCallError} carrying the classification, so the
140
282
  * caller can tell a refusal from an indeterminate drop.
283
+ *
284
+ * The rule is about the call as sent, not the operation's kind. A mutation
285
+ * the callee applies at most once per identity the call carries is
286
+ * repeatable by construction: a repeat of one that already ran is answered
287
+ * from the callee's record and applies nothing. Nimbus has two:
288
+ * - delivered filesystem mutations (@nimbus-sh/core supervisor-delivery):
289
+ * a delivery id plus the callee INSTANCE's incarnation. The record lives
290
+ * in that instance's memory, and any other instance — or a callee that
291
+ * predates delivery — refuses the call permanently rather than apply it
292
+ * without one;
293
+ * - appends: writer, module incarnation and operation sequence, recorded
294
+ * durably until acknowledged.
295
+ * Such a call goes through {@link idempotent}, re-sending the same identity
296
+ * on every attempt, with a `retryWindowMs` inside the callee's retention of
297
+ * that record. Without such an identity, a mutation stays here.
141
298
  */
142
299
  export async function mutating<S, T>(
143
300
  operation: string,