@nimbus-sh/fabric 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +487 -0
  3. package/dist/alarms.d.ts +134 -0
  4. package/dist/alarms.d.ts.map +1 -0
  5. package/dist/alarms.js +214 -0
  6. package/dist/bindings.d.ts +316 -0
  7. package/dist/bindings.d.ts.map +1 -0
  8. package/dist/bindings.js +678 -0
  9. package/dist/ctx-exports.d.ts +47 -0
  10. package/dist/ctx-exports.d.ts.map +1 -0
  11. package/dist/ctx-exports.js +54 -0
  12. package/dist/facet-image-store.d.ts +112 -0
  13. package/dist/facet-image-store.d.ts.map +1 -0
  14. package/dist/facet-image-store.js +181 -0
  15. package/dist/fanout-pool.d.ts +223 -0
  16. package/dist/fanout-pool.d.ts.map +1 -0
  17. package/dist/fanout-pool.js +368 -0
  18. package/dist/index.d.ts +26 -0
  19. package/dist/index.d.ts.map +1 -0
  20. package/dist/index.js +25 -0
  21. package/dist/inner-do-registry.d.ts +41 -0
  22. package/dist/inner-do-registry.d.ts.map +1 -0
  23. package/dist/inner-do-registry.js +51 -0
  24. package/dist/launch-journal.d.ts +170 -0
  25. package/dist/launch-journal.d.ts.map +1 -0
  26. package/dist/launch-journal.js +154 -0
  27. package/dist/launch-pacer.d.ts +173 -0
  28. package/dist/launch-pacer.d.ts.map +1 -0
  29. package/dist/launch-pacer.js +193 -0
  30. package/dist/loader-ledger.d.ts +57 -0
  31. package/dist/loader-ledger.d.ts.map +1 -0
  32. package/dist/loader-ledger.js +91 -0
  33. package/dist/loader-pool.d.ts +315 -0
  34. package/dist/loader-pool.d.ts.map +1 -0
  35. package/dist/loader-pool.js +666 -0
  36. package/dist/process-fabric.d.ts +524 -0
  37. package/dist/process-fabric.d.ts.map +1 -0
  38. package/dist/process-fabric.js +388 -0
  39. package/dist/process-host.d.ts +132 -0
  40. package/dist/process-host.d.ts.map +1 -0
  41. package/dist/process-host.js +444 -0
  42. package/dist/vendor/errors.d.ts +24 -0
  43. package/dist/vendor/errors.d.ts.map +1 -0
  44. package/dist/vendor/errors.js +46 -0
  45. package/dist/vendor/serialize.d.ts +3 -0
  46. package/dist/vendor/serialize.d.ts.map +1 -0
  47. package/dist/vendor/serialize.js +25 -0
  48. package/dist/vendor/types.d.ts +69 -0
  49. package/dist/vendor/types.d.ts.map +1 -0
  50. package/dist/vendor/types.js +4 -0
  51. package/dist/workerd-facet-host.d.ts +207 -0
  52. package/dist/workerd-facet-host.d.ts.map +1 -0
  53. package/dist/workerd-facet-host.js +508 -0
  54. package/dist/ws-hibernation-config.d.ts +73 -0
  55. package/dist/ws-hibernation-config.d.ts.map +1 -0
  56. package/dist/ws-hibernation-config.js +93 -0
  57. package/package.json +62 -0
  58. package/src/alarms.ts +275 -0
  59. package/src/bindings.ts +871 -0
  60. package/src/ctx-exports.ts +77 -0
  61. package/src/facet-image-store.ts +196 -0
  62. package/src/fanout-pool.ts +503 -0
  63. package/src/index.ts +26 -0
  64. package/src/inner-do-registry.ts +58 -0
  65. package/src/launch-journal.ts +229 -0
  66. package/src/launch-pacer.ts +231 -0
  67. package/src/loader-ledger.ts +112 -0
  68. package/src/loader-pool.ts +984 -0
  69. package/src/process-fabric.ts +729 -0
  70. package/src/process-host.ts +566 -0
  71. package/src/vendor/errors.ts +56 -0
  72. package/src/vendor/serialize.ts +37 -0
  73. package/src/vendor/types.ts +75 -0
  74. package/src/workerd-facet-host.ts +694 -0
  75. package/src/ws-hibernation-config.ts +123 -0
@@ -0,0 +1,229 @@
1
+ /**
2
+ * launch-journal.ts — durable record of the resident launches a Durable Object
3
+ * owes, and their recovery after an instance reset.
4
+ *
5
+ * The platform resets a session Durable Object over what one turn has
6
+ * outstanding in storage ("Internal error in Durable Object storage caused
7
+ * object to be reset"), and a resident launch is the largest writer a session
8
+ * has. Everything a launch holds is in memory, so the process it is building
9
+ * and the terminal watching it both go with the instance — the journal is what
10
+ * a LATER instance reads to know that happened, and this module is the whole
11
+ * of that mechanism: the put→sync durability barrier on the way in, the
12
+ * delete→sync release on the way out, and the once-per-instance recovery pump
13
+ * that re-drives what a previous generation left behind.
14
+ *
15
+ * What a launch IS stays the embedder's: the journal stores the record it is
16
+ * given and hands it back on recovery. The mechanism reads only the fields in
17
+ * {@link ResidentLaunchRecord}; everything else in the record rides through
18
+ * opaquely.
19
+ */
20
+
21
+ /**
22
+ * Prefix for the resident-process journal: one row per resident this session
23
+ * owes the user, keyed by the pid it was built for.
24
+ *
25
+ * A resident holds its state in memory — the process table entry, the facet
26
+ * handle, the terminal — so an instance reset destroys it silently. The row
27
+ * is what a LATER instance reads to know a resident ended that way rather
28
+ * than on purpose: a pid at or below the reader's own pid base was allocated
29
+ * by a previous generation (PID_GEN_STRIDE, core's process-table). Written
30
+ * (and synced) before the launch's first byte of work, rewritten as `running`
31
+ * when the launch settles, and released only when the PROCESS ends — because
32
+ * the resets this row survives strike after the launch as often as during it
33
+ * (measured live, staging 2026-08-13: every observed reset landed seconds
34
+ * AFTER settle).
35
+ *
36
+ * The VALUE is live production DO storage and must never change — renaming a
37
+ * storage key is a migration, and orphaned rows are the least of what it
38
+ * breaks.
39
+ */
40
+ export const RESIDENT_LAUNCH_KEY_PREFIX = 'resident-launch:';
41
+
42
+ /** A launch is re-driven once. A reset that recurs is not the transient one. */
43
+ export const RESIDENT_LAUNCH_MAX_ATTEMPT = 1;
44
+
45
+ /**
46
+ * A resident process this session owes the user, as a later instance would
47
+ * have to re-drive it.
48
+ *
49
+ * The launch's own inputs and nothing derived from them: everything a launch
50
+ * builds is a pure function of these, and the images it writes are content-
51
+ * addressed, so re-driving is the same work again rather than a repair. The
52
+ * inputs themselves are the embedder's — a record type extends this base with
53
+ * whatever its `redrive` needs, and the journal never reads those fields.
54
+ *
55
+ * The row lives for the PROCESS's lifetime, not the launch's. Measured live
56
+ * (staging, 2026-08-13): every observed reset struck seconds AFTER the launch
57
+ * settled — the platform kills the object while the resident runs, which is
58
+ * when a launch-scoped row had already been deleted and recovery had nothing
59
+ * to find. A resident's facet cannot outlive its session instance (the
60
+ * process host's held-open leg dies with it), so a row from a previous
61
+ * generation always names a process that is genuinely gone.
62
+ */
63
+ export interface ResidentLaunchRecord {
64
+ pid: number;
65
+ command: string;
66
+ /** 0 for a launch the user asked for; 1 for the one re-drive it may get. */
67
+ attempt: number;
68
+ /** Where the resident was when its instance died: still being built, or
69
+ * booted and running. Running residents re-drive with a fresh attempt
70
+ * budget — their launch already proved itself once. */
71
+ phase: 'starting' | 'running';
72
+ }
73
+
74
+ /**
75
+ * The slice of Durable Object storage the journal writes through. Exactly a
76
+ * `DurableObjectStorage`, narrowed to what the mechanism performs — `sync()`
77
+ * is load-bearing, see {@link ResidentLaunchJournal.journal}.
78
+ */
79
+ export interface LaunchJournalStorage {
80
+ put(key: string, value: unknown): Promise<void>;
81
+ delete(key: string): Promise<boolean>;
82
+ list<T = unknown>(options: { prefix: string }): Promise<Map<string, T>>;
83
+ sync(): Promise<void>;
84
+ }
85
+
86
+ /** What the journal's recovery needs from its embedder. */
87
+ export interface LaunchJournalHost<R extends ResidentLaunchRecord> {
88
+ /**
89
+ * The current instance generation's pid floor. A pid at or below it was
90
+ * allocated by a PREVIOUS instance (core's process-table, PID_GEN_STRIDE),
91
+ * so its launch never finished; above it is this instance's own, still
92
+ * running. The journal takes the base rather than the predicate so the one
93
+ * definition of what a prior-generation pid is stays in the process table.
94
+ */
95
+ generationBase(): number;
96
+ /**
97
+ * Root a recovery re-drive on the instance (`ctx.waitUntil`) so it is not
98
+ * an abandoned promise between turns.
99
+ */
100
+ waitUntil(promise: Promise<unknown>): void;
101
+ /**
102
+ * Re-drive an interrupted launch from its journalled inputs. `attempt` is
103
+ * the budget the re-drive spends — the mechanism computes it, the embedder
104
+ * carries it into the launch it starts. The result is discarded: a re-drive
105
+ * owns its own process, and nobody is waiting on the pid it allocates.
106
+ */
107
+ redrive(record: R, attempt: number): Promise<unknown>;
108
+ /** A re-drive is being started for this record. */
109
+ onRedrive?(record: R): void;
110
+ /** The record's re-drive budget is spent; the resident stays stopped. */
111
+ onAbandoned?(record: R): void;
112
+ /** The re-drive itself failed. */
113
+ onRedriveFailed?(record: R, error: unknown): void;
114
+ }
115
+
116
+ /**
117
+ * The resident-launch journal of one Durable Object instance.
118
+ *
119
+ * In-memory state here is per-instance on purpose: `journalledPids` tracks the
120
+ * rows THIS instance wrote, and `recovered` whether this instance has already
121
+ * read the journal a reset leaves behind. Rows from a previous instance are
122
+ * recovery's to consume, never the release path's.
123
+ */
124
+ export class ResidentLaunchJournal<R extends ResidentLaunchRecord> {
125
+ /**
126
+ * Pids THIS instance holds journal rows for. What keeps the terminal hook —
127
+ * which fires for every process, shells and one-shots included — from
128
+ * paying a storage delete for pids that never had a row.
129
+ */
130
+ private journalledPids = new Set<number>();
131
+ /** Whether this instance has already read the journal a reset leaves behind. */
132
+ private recovered = false;
133
+
134
+ constructor(
135
+ private readonly storage: LaunchJournalStorage,
136
+ private readonly host: LaunchJournalHost<R>,
137
+ ) {}
138
+
139
+ /**
140
+ * Record a launch as in flight, so an instance that replaces this one knows
141
+ * it never finished. Best-effort: a launch that cannot be journalled still
142
+ * runs, and a reset then costs exactly what it cost before the journal.
143
+ *
144
+ * Synced, not merely put: `await put()` resolves before durability, and the
145
+ * reset this journal exists for destroys every write its turn still had
146
+ * outstanding — measured live, a launch killed in its first chunks left NO
147
+ * row for the replacement instance to find, which is how the recovery this
148
+ * feeds sat inert while its own test stayed green. `sync()` is the storage
149
+ * layer's durability barrier: the row is on disk before the launch performs
150
+ * its first byte of real work. What remains is a reset between the put and
151
+ * the sync's completion — and a launch that dies there has not started, so
152
+ * losing its row costs a retype, not a recovery.
153
+ */
154
+ async journal(record: R): Promise<void> {
155
+ try {
156
+ this.journalledPids.add(record.pid);
157
+ await this.storage.put(`${RESIDENT_LAUNCH_KEY_PREFIX}${record.pid}`, record);
158
+ await this.storage.sync();
159
+ } catch (e: unknown) {
160
+ console.warn('[nimbus] resident launch journal write failed:', errorMessage(e));
161
+ }
162
+ }
163
+
164
+ /** True while this instance holds a journal row for `pid`. */
165
+ has(pid: number): boolean {
166
+ return this.journalledPids.has(pid);
167
+ }
168
+
169
+ /**
170
+ * The journal row's one release: the process is over, nothing is owed.
171
+ * Synced so an instance reset moments later cannot roll the delete back and
172
+ * resurrect a process the user watched end.
173
+ */
174
+ async release(pid: number): Promise<void> {
175
+ if (!this.journalledPids.delete(pid)) return;
176
+ try {
177
+ await this.storage.delete(`${RESIDENT_LAUNCH_KEY_PREFIX}${pid}`);
178
+ await this.storage.sync();
179
+ } catch (e: unknown) {
180
+ console.warn('[nimbus] resident launch journal delete failed:', errorMessage(e));
181
+ }
182
+ }
183
+
184
+ /**
185
+ * Re-drive the launches a previous instance was building when it was reset.
186
+ *
187
+ * Sited on the launch-turn pump because the pump is what an alarm calls, and
188
+ * a launch that was suspended has an alarm armed for it — a reset during a
189
+ * chunk fails that alarm, and the platform re-delivers it to the instance
190
+ * that replaces this one. So the first turn after a reset is already this
191
+ * one.
192
+ *
193
+ * Runs once per instance: the journal only changes when a launch of THIS
194
+ * instance starts or settles, and those are rows this instance wrote.
195
+ */
196
+ async recoverInterrupted(): Promise<void> {
197
+ if (this.recovered) return;
198
+ this.recovered = true;
199
+ const journal = await this.storage.list<R>({ prefix: RESIDENT_LAUNCH_KEY_PREFIX });
200
+ const base = this.host.generationBase();
201
+ for (const [key, record] of journal) {
202
+ // A pid at or below this instance's base was allocated by a PREVIOUS one
203
+ // (process-table.ts, PID_GEN_STRIDE), so its launch never finished; above
204
+ // the base is this instance's own, still running. Same predicate as
205
+ // `session/rpc.ts` uses to attribute a prior generation's pid.
206
+ if (!(record.pid > 0 && record.pid <= base)) continue;
207
+ await this.storage.delete(key);
208
+ if (record.attempt >= RESIDENT_LAUNCH_MAX_ATTEMPT) {
209
+ this.host.onAbandoned?.(record);
210
+ continue;
211
+ }
212
+ this.host.onRedrive?.(record);
213
+ // Not awaited: this call is running inside the alarm that granted the
214
+ // turn, and the launch it starts asks for turns of its own through that
215
+ // same alarm — awaiting it here would be waiting on an alarm that cannot
216
+ // be scheduled until this one returns.
217
+ this.host.waitUntil(
218
+ this.host.redrive(record, record.attempt + 1)
219
+ .catch((e: unknown) => {
220
+ this.host.onRedriveFailed?.(record, e);
221
+ }),
222
+ );
223
+ }
224
+ }
225
+ }
226
+
227
+ function errorMessage(error: unknown): string {
228
+ return error instanceof Error ? error.message : String(error);
229
+ }
@@ -0,0 +1,231 @@
1
+ /**
2
+ * launch-pacer.ts — spreading a resident launch across Durable Object turns.
3
+ *
4
+ * Building a resident process is the largest single span of computation this
5
+ * session performs: for pi it walks a 17 MB source tree through eight
6
+ * enrichment passes, serializes a 22.9 MB module map, and writes that map into
7
+ * the image store. Done in one turn it occupied the session DO's only thread
8
+ * for 15-35 s, and a session that cannot reach its thread cannot service the
9
+ * terminal WebSocket — the launch turn finished `outcome=ok` and the terminal
10
+ * died anyway, painting "[process terminal closed]" over a process that was
11
+ * still running.
12
+ *
13
+ * A faster launch does not fix that. A launch half the length still blocks the
14
+ * thread for as long as it runs, and the socket is dropped inside that window
15
+ * whether or not the work succeeds. What fixes it is never holding the thread
16
+ * for long in the first place, which means suspending the launch at bounded
17
+ * intervals and resuming it on a fresh turn. Responsiveness stops depending on
18
+ * how long the total work takes.
19
+ *
20
+ * A fresh turn is also a fresh CPU budget. The same launches that dropped the
21
+ * socket were also being killed with `exceededCpu` at 31.8 s and 32.5 s
22
+ * against a 30 s ceiling, and no amount of yielding *within* one invocation
23
+ * moves that: CPU accrues to the invocation, not to the pause. Only genuinely
24
+ * re-entering the object resets it.
25
+ *
26
+ * Progress is measured in bytes rather than milliseconds because workerd's
27
+ * clock does not advance without I/O — a wall-clock guard inside a span of
28
+ * pure computation reads zero however many seconds it burns, which is why the
29
+ * phase costs behind this module had to be recovered from per-turn `cpuTime`
30
+ * rather than measured in place. Bytes are what the work is actually
31
+ * proportional to, and they are exact. The same reasoning is why
32
+ * `git/network-facet.ts` bounds its checkout chunks by entries and decoded
33
+ * bytes and treats its wall guard as coarse.
34
+ */
35
+
36
+ /** How a paced launch gets back onto a fresh Durable Object turn. */
37
+ export interface LaunchTurnScheduler {
38
+ /**
39
+ * Suspend until a fresh turn is running this launch again.
40
+ *
41
+ * `chunkEnded` settles when the resumed launch reaches its next suspension
42
+ * point or finishes, so whoever grants the turn can await the work it just
43
+ * released rather than letting it run detached in a handler's microtask
44
+ * drain.
45
+ */
46
+ nextTurn(chunkEnded: Promise<void>): Promise<void>;
47
+ }
48
+
49
+ /**
50
+ * Bytes of launch work one turn may perform before it must yield.
51
+ *
52
+ * Sized so a chunk stays far below both the CPU ceiling and the span in which
53
+ * a terminal socket is at risk, while keeping the number of turn handoffs —
54
+ * each an alarm round trip — small enough not to dominate a launch. pi's
55
+ * 22.9 MB map crosses this about a dozen times per phase that handles it.
56
+ */
57
+ export const LAUNCH_CHUNK_MAX_BYTES = 2_000_000;
58
+
59
+ /**
60
+ * Accounts launch progress and ends the turn when a chunk's worth has been
61
+ * spent.
62
+ *
63
+ * Callers report the work they are about to do or have just done and await
64
+ * the result; a pacer that is not yielding returns without suspending, so the
65
+ * one-shot exec path — which passes no pacer at all — keeps its exact
66
+ * behaviour and cost. Nothing here decides WHAT the launch does, only where it
67
+ * is allowed to stop.
68
+ */
69
+ export class LaunchPacer {
70
+ /** Turn handoffs this launch has taken. Reported with the launch. */
71
+ chunks = 0;
72
+ /** Total work accounted, for the same report. */
73
+ bytes = 0;
74
+
75
+ private spent = 0;
76
+ private chunkEnded: { promise: Promise<void>; resolve: () => void } | undefined;
77
+
78
+ /**
79
+ * @param stillWanted Checked every time the launch resumes. A launch spans
80
+ * many turns, so anything may have happened to what it is building for
81
+ * while it was suspended; throwing from here is how a launch stops instead
82
+ * of spending turn after turn on work nothing will use. Checked at the one
83
+ * place a launch can be interrupted, rather than at whichever phases
84
+ * remembered to ask.
85
+ */
86
+ constructor(
87
+ private readonly scheduler: LaunchTurnScheduler,
88
+ private readonly maxChunkBytes: number = LAUNCH_CHUNK_MAX_BYTES,
89
+ private readonly stillWanted?: () => void,
90
+ ) {}
91
+
92
+ /**
93
+ * Account `bytes` of completed work, ending the turn if a chunk is full.
94
+ *
95
+ * Safe to call anywhere the launch holds no state that a concurrent turn
96
+ * could invalidate — which is why the image store registers its whole root
97
+ * set before the first call rather than one entry at a time.
98
+ */
99
+ async spend(bytes: number): Promise<void> {
100
+ this.bytes += bytes;
101
+ this.spent += bytes;
102
+ if (this.spent < this.maxChunkBytes) return;
103
+ this.spent = 0;
104
+ this.chunks++;
105
+ // Release the turn that resumed us before asking for the next one.
106
+ this.chunkEnded?.resolve();
107
+ const ended = withResolvers();
108
+ this.chunkEnded = ended;
109
+ await this.scheduler.nextTurn(ended.promise);
110
+ this.stillWanted?.();
111
+ }
112
+
113
+ /**
114
+ * The launch has finished (or failed). Releases the turn still waiting on
115
+ * the chunk it resumed, so a launch that ends mid-chunk does not strand the
116
+ * handler that granted it.
117
+ */
118
+ settle(): void {
119
+ this.chunkEnded?.resolve();
120
+ this.chunkEnded = undefined;
121
+ }
122
+ }
123
+
124
+ function withResolvers(): { promise: Promise<void>; resolve: () => void } {
125
+ let resolve!: () => void;
126
+ const promise = new Promise<void>((r) => { resolve = r; });
127
+ return { promise, resolve };
128
+ }
129
+
130
+ /** What {@link LaunchTurnPump} needs from the Durable Object hosting it. */
131
+ export interface LaunchTurnPumpHost {
132
+ /**
133
+ * Arrange for {@link LaunchTurnPump.pump} to run on a fresh Durable Object
134
+ * turn.
135
+ *
136
+ * The embedder satisfies this with an alarm, which is the only primitive
137
+ * that genuinely re-enters the object: a fresh turn is both a released
138
+ * thread and a fresh CPU budget, and a launch needs each for a different
139
+ * reason. Without it the pump degrades to a same-context timer — see
140
+ * {@link LaunchTurnPump.nextTurn}.
141
+ */
142
+ requestTurn?: () => void;
143
+ /**
144
+ * Awaited first on every pump, before any waiter resumes. Where the
145
+ * resident-launch journal's recovery sits: the pump is what an alarm calls,
146
+ * and the first turn after a reset is the re-delivered alarm of a launch
147
+ * the reset interrupted.
148
+ */
149
+ recover?: () => Promise<void>;
150
+ }
151
+
152
+ /**
153
+ * The granting side of {@link LaunchTurnScheduler}: parks suspended launches
154
+ * and resumes every one of them when the host grants a fresh turn.
155
+ */
156
+ export class LaunchTurnPump implements LaunchTurnScheduler {
157
+ /**
158
+ * Launches suspended between chunks, waiting for a turn of their own.
159
+ *
160
+ * In-memory on purpose: a launch is only meaningful while the process table
161
+ * entry it is building for exists, and both are lost together if the isolate
162
+ * resets. What survives a reset is the journal, which names the launch's
163
+ * INPUTS rather than its position — a resumed queue would be resurrecting
164
+ * half-built work for pids that no longer exist, where re-driving a launch
165
+ * from its inputs is the same idempotent work again.
166
+ */
167
+ private waiters: Array<{ resume: () => void; chunkEnded: Promise<void> }> = [];
168
+
169
+ constructor(private readonly host: LaunchTurnPumpHost) {}
170
+
171
+ /**
172
+ * How a paced launch asks for a fresh turn.
173
+ *
174
+ * The host grants one by calling {@link pump} from a context that is
175
+ * genuinely a new invocation — the session's alarm. Without such a host
176
+ * there is no fresh turn to be had, and the launch continues on this one
177
+ * rather than hanging: that is exactly the single-turn launch this path has
178
+ * always performed, so a harness or a runtime without alarms loses the
179
+ * responsiveness but keeps the behaviour.
180
+ */
181
+ nextTurn(chunkEnded: Promise<void>): Promise<void> {
182
+ return new Promise<void>((resume) => {
183
+ this.waiters.push({ resume, chunkEnded });
184
+ if (this.host.requestTurn) {
185
+ this.host.requestTurn();
186
+ return;
187
+ }
188
+ setTimeout(() => { void this.pump(); }, 0);
189
+ });
190
+ }
191
+
192
+ /**
193
+ * Run one chunk of every launch waiting for a turn.
194
+ *
195
+ * Awaits the chunk each resumed launch then performs, so the invocation that
196
+ * granted the turn is the invocation that pays for the work — rather than
197
+ * releasing it into a handler's microtask drain, where nothing owns it and
198
+ * the runtime may tear the context down mid-chunk.
199
+ */
200
+ async pump(): Promise<void> {
201
+ await this.host.recover?.();
202
+ const waiting = this.waiters;
203
+ if (waiting.length === 0) return;
204
+ this.waiters = [];
205
+ for (const waiter of waiting) waiter.resume();
206
+ await Promise.all(waiting.map((waiter) => waiter.chunkEnded));
207
+ }
208
+
209
+ /** Whether any launch is suspended waiting for a turn. */
210
+ get hasPending(): boolean {
211
+ return this.waiters.length > 0;
212
+ }
213
+ }
214
+
215
+ /**
216
+ * Chunk bound for this session, honouring the verification knob.
217
+ *
218
+ * `NIMBUS_LAUNCH_CHUNK_BYTES` forces a small bound so an ordinary launch —
219
+ * not just a pathological one — crosses several turns and exercises every
220
+ * suspension point. Without it the multi-turn path would only ever be
221
+ * reached by the largest programs, which is the same reason
222
+ * `git/commands.ts` carries `NIMBUS_GIT_CHECKOUT_CHUNK_ENTRIES`. Unset in
223
+ * production, where the default applies.
224
+ */
225
+ export function launchChunkMaxBytes(env: unknown): number {
226
+ const raw = (env as { NIMBUS_LAUNCH_CHUNK_BYTES?: string } | null | undefined)
227
+ ?.NIMBUS_LAUNCH_CHUNK_BYTES;
228
+ if (!raw) return LAUNCH_CHUNK_MAX_BYTES;
229
+ const parsed = Number(raw);
230
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : LAUNCH_CHUNK_MAX_BYTES;
231
+ }
@@ -0,0 +1,112 @@
1
+ /**
2
+ * loader-ledger.ts — per-DO accounting for the Worker Loader's two caps.
3
+ *
4
+ * Measured on production workerd: a Durable Object admits ~5–6 concurrent
5
+ * dynamic workers before the platform refuses with "Too many concurrent
6
+ * dynamic workers", one DO method can drive at most 4 concurrent Loader
7
+ * fetches, and loader-cache entries are never released — every DISTINCT
8
+ * `loader.get(id)` permanently consumes one of the dynamic-worker slots for
9
+ * the object's lifetime. Nimbus stays under the caps by construction
10
+ * (`IN_DO_THRESHOLD` = 5 in the fanout pool), which until now meant the slots
11
+ * were counted in prose. This ledger counts them at the fabric's loader call
12
+ * sites instead — the loader pool's slots, a resident process's keyed worker,
13
+ * a one-shot's load — so proximity is measurable and a cap failure can name
14
+ * the ids actually holding slots.
15
+ *
16
+ * Measurement only: no admission control. The caps are the platform's, they
17
+ * are approximate ("~5–6"), and a gate on an approximate number would refuse
18
+ * work the platform would have run.
19
+ *
20
+ * Keyed weakly off the hosting actor's `ctx`, like the facet slot books: the
21
+ * caps are per Durable Object, and dynamic workers die with the isolate that
22
+ * loaded them, so a ledger that goes away with its host describes nothing
23
+ * that still exists.
24
+ */
25
+
26
+ import { classifyError } from '@nimbus-sh/core/observability/oom-classify.js';
27
+
28
+ interface LoaderLedger {
29
+ /** Distinct loader ids ever gotten — each one a permanently consumed slot. */
30
+ ids: Set<string>;
31
+ /** In-flight calls into dynamic workers, right now. */
32
+ liveFetches: number;
33
+ /** The most that were ever in flight at once. */
34
+ peakLiveFetches: number;
35
+ }
36
+
37
+ const ledgers = new WeakMap<object, LoaderLedger>();
38
+
39
+ function ledger(ctx: object): LoaderLedger {
40
+ let entry = ledgers.get(ctx);
41
+ if (!entry) {
42
+ entry = { ids: new Set(), liveFetches: 0, peakLiveFetches: 0 };
43
+ ledgers.set(ctx, entry);
44
+ }
45
+ return entry;
46
+ }
47
+
48
+ /** Record a keyed `loader.get(id)` — a permanent slot if the id is new. */
49
+ export function recordLoaderId(ctx: object, id: string): void {
50
+ ledger(ctx).ids.add(id);
51
+ }
52
+
53
+ /**
54
+ * Count one call into a dynamic worker as a live Loader fetch; the returned
55
+ * function ends it (idempotently), from the caller's own `finally`.
56
+ *
57
+ * A begin/end pair rather than a wrapper on purpose, and the shape is
58
+ * load-bearing: wrapping the stub call in a ledger-owned async frame
59
+ * (`trackLoaderFetch(ctx, () => entrypoint.execute(...))`) left the hosting
60
+ * Durable Object poisoned after every pooled dispatch — the next fabric
61
+ * activity hung the object or reset the instance outright (pid base jumped,
62
+ * every attached WebSocket dropped with no close frame), measured 7/7 on
63
+ * staging and gone 3/3 with the direct call restored. Same seam-quirk class
64
+ * as pipelined `fetch.call`, which workerd refuses for dynamically-loaded
65
+ * workers: an RPC stub call must stay a direct property call awaited by the
66
+ * frame that made it, so the ledger only brackets it.
67
+ */
68
+ export function beginLoaderFetch(ctx: object): () => void {
69
+ const entry = ledger(ctx);
70
+ entry.liveFetches++;
71
+ entry.peakLiveFetches = Math.max(entry.peakLiveFetches, entry.liveFetches);
72
+ let ended = false;
73
+ return () => {
74
+ if (ended) return;
75
+ ended = true;
76
+ entry.liveFetches--;
77
+ };
78
+ }
79
+
80
+ /** Snapshot for the diag surface. Pure read; no I/O. */
81
+ export function loaderLedgerStats(ctx: object): {
82
+ idsEverGotten: string[];
83
+ liveFetches: number;
84
+ peakLiveFetches: number;
85
+ } {
86
+ const entry = ledger(ctx);
87
+ return {
88
+ idsEverGotten: [...entry.ids],
89
+ liveFetches: entry.liveFetches,
90
+ peakLiveFetches: entry.peakLiveFetches,
91
+ };
92
+ }
93
+
94
+ /**
95
+ * Name the per-DO accounting on a "Too many concurrent dynamic workers"
96
+ * failure; hand every other error back untouched. The platform's message
97
+ * says only that the cap was hit — which ids hold the slots, and that a
98
+ * keyed id can never give one back, is what the operator needs to know to
99
+ * shrink anything.
100
+ */
101
+ export function withDynamicWorkerCapNamed<E>(ctx: object, error: E): E | Error {
102
+ if (classifyError(error) !== 'dynamic_worker_cap') return error;
103
+ const entry = ledger(ctx);
104
+ const platform = error instanceof Error ? error.message : String(error);
105
+ return new Error(
106
+ `${platform} — this Durable Object has ${entry.ids.size} loader id(s) permanently `
107
+ + `holding dynamic-worker slots (a loader.get id is never released): `
108
+ + `${[...entry.ids].join(', ') || '(none recorded)'}; live Loader fetches ${entry.liveFetches}, `
109
+ + `peak ${entry.peakLiveFetches}`,
110
+ { cause: error },
111
+ );
112
+ }