@effect-agent/platform-cloudflare 0.1.0-beta.111 → 0.1.0-beta.113

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/Alarm.mjs CHANGED
@@ -2,11 +2,12 @@ import { t as __exportAll } from "./rolldown-runtime-D7D4PA-g.mjs";
2
2
  import { DurableObjectContext } from "./CloudflareBindings.mjs";
3
3
  import { AuxiliaryDispatchMillis, CloudflareDurableRuntimeConfig } from "./CloudflareConfig.mjs";
4
4
  import { r as safeCauseMessage } from "./boundary-BguazVkh.mjs";
5
- import { Cause, Clock, Context, DateTime, Deferred, Effect, Exit, Fiber, Layer, Option, Random, Ref, Schema, Scope, Semaphore, Stream, Struct } from "effect";
5
+ import { Cause, Clock, Context, DateTime, Deferred, Effect, Exit, Fiber, Layer, Option, PubSub, Random, Ref, Result, Schema, Scope, Semaphore, Stream, Struct } from "effect";
6
6
  import "effect-agent/agent-registration";
7
- import { DurableAgentRuntime } from "effect-agent/durable-agent-runtime";
7
+ import { DurableAgentRuntime, DurableRuntimeConfig, RecoveryBlocked, RecoveryFailure, RecoverySweepResult, isolateRecovery } from "effect-agent/durable-agent-runtime";
8
8
  import { SubmissionId, ThreadId } from "effect-agent/identifiers";
9
- import { SubmissionLedger } from "effect-agent/submission-ledger";
9
+ import { OperationAuthorizationRequest, OperationAuthorizer } from "effect-agent/operation-authorizer";
10
+ import { AbortIntentRequest, SubmissionLedger } from "effect-agent/submission-ledger";
10
11
  import { ThreadProjectionMaintenance, drainDue } from "effect-agent/thread-projection-maintenance";
11
12
  import { WakeScheduler } from "effect-agent/wake-scheduler";
12
13
  import { SqlClient } from "effect/unstable/sql/SqlClient";
@@ -17,10 +18,12 @@ var Alarm_exports = /* @__PURE__ */ __exportAll({
17
18
  MaintenancePassReport: () => MaintenancePassReport,
18
19
  ThreadHostMaintenance: () => ThreadHostMaintenance,
19
20
  ThreadMaintenance: () => ThreadMaintenance,
21
+ ThreadMaintenanceActivity: () => ThreadMaintenanceActivity,
20
22
  ThreadMaintenanceFailpoint: () => ThreadMaintenanceFailpoint,
21
23
  ThreadMessageDelivery: () => ThreadMessageDelivery,
22
24
  ThreadMutationGate: () => ThreadMutationGate,
23
25
  ThreadPublication: () => ThreadPublication,
26
+ ThreadRecoveryFault: () => ThreadRecoveryFault,
24
27
  publishCommitted: () => publishCommitted
25
28
  });
26
29
  /**
@@ -87,7 +90,7 @@ var DurableAlarmService = class DurableAlarmService extends Context.Service()("@
87
90
  var MaintenancePassReport = class extends Schema.Class("@effect-agent/platform-cloudflare/MaintenancePassReport")({
88
91
  /** `caught-up` ran no runtime work (publication may be pending); `actionable` ran recovery. */
89
92
  phase: Schema.Literals(["caught-up", "actionable"]),
90
- /** Recovery decisions executed (or deferred) BEFORE any new claim in this pass. */
93
+ /** Recovery decisions and Thread faults observed during this event. */
91
94
  recovered: Schema.Int.check(Schema.isGreaterThanOrEqualTo(0)),
92
95
  /** Head Attempts settled during the event. Joined input may settle with each head. */
93
96
  settled: Schema.Int.check(Schema.isGreaterThanOrEqualTo(0)),
@@ -109,6 +112,27 @@ var ThreadPublication = class extends Context.Service()("@effect-agent/platform-
109
112
  pendingDeadline: Effect.succeed(Option.none())
110
113
  });
111
114
  };
115
+ /** Event-local accounting shared by the existing maintenance pumps and native scheduler. */
116
+ var ThreadMaintenanceActivity = class ThreadMaintenanceActivity extends Context.Service()("@effect-agent/platform-cloudflare/ThreadMaintenanceActivity") {
117
+ /** Join child pumps; acknowledge the parent only after every child is ready or has exited. */
118
+ static all = Effect.fnUntraced(function* (lanes) {
119
+ const activity = yield* ThreadMaintenanceActivity;
120
+ let remaining = lanes.length;
121
+ if (remaining === 0) return yield* activity.ready.pipe(Effect.as([]));
122
+ return yield* Effect.forEach(lanes, (lane) => {
123
+ let ready = false;
124
+ const child = {
125
+ ...activity,
126
+ ready: Effect.suspend(() => {
127
+ if (ready) return Effect.void;
128
+ ready = true;
129
+ return --remaining === 0 ? activity.ready : Effect.void;
130
+ }).pipe(Effect.uninterruptible)
131
+ };
132
+ return lane.pipe(Effect.provideService(ThreadMaintenanceActivity, child), Effect.onExit(() => child.ready));
133
+ }, { concurrency: "unbounded" });
134
+ });
135
+ };
112
136
  /**
113
137
  * Host-assembled native message recovery. The driver bounds each actual Claim and persists its
114
138
  * timeout/retry before this pump returns. Do not add a second timer starting at batch selection:
@@ -120,14 +144,16 @@ const ThreadMessageDelivery = Context.Reference("@effect-agent/platform-cloudfla
120
144
  }) });
121
145
  /**
122
146
  * Application obligations sharing this Object's alarm. Admit one initial external wave even on
123
- * a caught-up pass, then respond to wakes until dispatchClosed. This closes only admission
124
- * of NEW external waves; native work can continue while admitted waves finish. Keep local
125
- * admission/hub subscriptions in the event Scope until maintenance tears it down.
147
+ * a caught-up pass, then respond to wakes and ThreadMaintenanceActivity.subscribeChanges until
148
+ * dispatchClosed. Yield the event's ThreadMaintenanceActivity to bracket each finite lane with
149
+ * run and acknowledge initial setup with ready.
150
+ * Passive subscriptions do not count as active work. Closure stops both new waves and native
151
+ * Attempts; local admission/control subscriptions belong to the event Scope until teardown.
126
152
  * No deadline sleeps or automatic retry loops. Return after already-admitted waves finish.
127
153
  *
128
154
  * Declare a finite whole-wave allowance (1..300000ms): maximum for parallel lanes, sum for
129
155
  * sequential operations. Admit a wave only if its allowance fits before dispatchUntil. Later
130
- * arrivals cannot renew the retirement window. Maintenance bounds the join after dispatchClosed,
156
+ * arrivals cannot renew an admitted wave's allowance. Maintenance bounds each registered wave,
131
157
  * interrupts and joins event Scope, then reads local deadlines under the mutation gate.
132
158
  *
133
159
  * Setup and pendingDeadline are bounded local operations. Persist claims/envelopes before
@@ -155,6 +181,30 @@ var BindingRetry = class extends Schema.Class("BindingRetry")({
155
181
  notBefore: Schema.Finite,
156
182
  reportedAt: Schema.Finite
157
183
  }) {};
184
+ /**
185
+ * A durable observation of blocked recovery, independent of the execution journal. This is
186
+ * neither a Settlement nor proof that external effects did not happen. A successful recovery
187
+ * sweep clears it; repair must preserve canonical history and the original admission identity.
188
+ */
189
+ var ThreadRecoveryFault = class extends Schema.Class("@effect-agent/platform-cloudflare/ThreadRecoveryFault")({
190
+ schemaVersion: Schema.Literal(1),
191
+ threadId: ThreadId,
192
+ firstFailedAt: Schema.Finite,
193
+ lastFailedAt: Schema.Finite,
194
+ /** Earliest automatic recovery retry; new admissions do not erase this deadline. */
195
+ retryAt: Schema.Finite,
196
+ /** Saturates at 2^31 - 1; one observation per Thread per recovery sweep. */
197
+ attempts: Schema.Int.check(Schema.isGreaterThan(0), Schema.isLessThanOrEqualTo(2147483647)),
198
+ failure: RecoveryFailure
199
+ }) {};
200
+ const recoveryFaultKey = (threadId) => `effect-agent:thread-recovery-fault:v1:${threadId}`;
201
+ const decodeRecoveryFaultValue = Schema.decodeUnknownSync(ThreadRecoveryFault);
202
+ const decodeRecoveryFault = (threadId, encoded) => {
203
+ const fault = decodeRecoveryFaultValue(encoded);
204
+ if (fault.threadId !== threadId) throw new Error("Recovery status does not match its Thread key");
205
+ return fault;
206
+ };
207
+ const encodeRecoveryFault = Schema.encodeSync(ThreadRecoveryFault);
158
208
  var MaintenanceRetry = class extends Schema.Class("MaintenanceRetry")({
159
209
  generation: MaintenanceGeneration,
160
210
  notBefore: Schema.Finite,
@@ -169,6 +219,8 @@ var ThreadMaintenanceState = class extends Schema.Class("@effect-agent/platform-
169
219
  nonterminal: Schema.Int.check(Schema.isGreaterThanOrEqualTo(0)),
170
220
  /** One physical-owner cursor; old single-lane records need no conversion. */
171
221
  lastServedThreadId: Schema.optionalKey(ThreadId),
222
+ /** Rotate old recovery independently of dispatch, including after eviction or timeout. */
223
+ lastRecoveredThreadId: Schema.optionalKey(ThreadId),
172
224
  bindingRetries: Schema.optionalKey(Schema.Array(BindingRetry)),
173
225
  /** Absent on older records. A newer mutation makes this retry obsolete. */
174
226
  retry: Schema.optionalKey(MaintenanceRetry)
@@ -258,10 +310,10 @@ var ThreadMutationGate = class ThreadMutationGate extends Context.Service()("@ef
258
310
  * recovery, ledger scans or canonical-history reads.
259
311
  * 2. Reconcile before each head Attempt, then checkpoint only the observed generation. A racing
260
312
  * producer keeps its newer generation dirty. Native retries retain their durable backoff.
261
- * 3. After the initial native opportunity, stop admitting new external waves and let the
262
- * active waves finish. While they remain in flight, native wakes and bounded scans can
263
- * advance more heads. All Attempts share the event's original ten-minute yield deadline.
264
- * 4. Native message delivery retains its driver-owned Claim deadline. Host/backfill joins are
313
+ * 3. Keep native and delivery admission open together while finite waves remain active, so
314
+ * fresh replies and abort controls can progress during unrelated cleanup. Close atomically
315
+ * at quiescence, or at the original ten-minute yield deadline, before retiring listeners.
316
+ * 4. Native message delivery retains its driver-owned Claim deadline. Host/backfill waves are
265
317
  * bounded independently; incoming native work never restarts or cancels their attempts.
266
318
  * Auxiliary failures are reported after the current native opportunity.
267
319
  * 5. Close every event resource before the final gated deadline snapshot and alarm decision.
@@ -270,6 +322,7 @@ var ThreadMutationGate = class ThreadMutationGate extends Context.Service()("@ef
270
322
  var ThreadMaintenance = class ThreadMaintenance extends Context.Service()("@effect-agent/platform-cloudflare/ThreadMaintenance") {
271
323
  static layer = Layer.effect(ThreadMaintenance)(Effect.gen(function* () {
272
324
  const runtime = yield* DurableAgentRuntime;
325
+ const recoveryConfig = yield* DurableRuntimeConfig;
273
326
  const ledger = yield* SubmissionLedger;
274
327
  const wakes = yield* WakeScheduler;
275
328
  const alarm = yield* DurableAlarmService;
@@ -281,6 +334,7 @@ var ThreadMaintenance = class ThreadMaintenance extends Context.Service()("@effe
281
334
  const projection = yield* ThreadProjectionMaintenance;
282
335
  const messages = yield* ThreadMessageDelivery;
283
336
  const host = yield* ThreadHostMaintenance;
337
+ const authorizer = yield* OperationAuthorizer;
284
338
  const projectionDeadline = projection.pendingDeadline.pipe(Effect.catchCauseIf((cause) => !Cause.hasInterrupts(cause), (cause) => Effect.logError("Thread projection deadline unavailable", cause).pipe(Effect.as(Option.some(0)))));
285
339
  const pendingDeadline = Effect.gen(function* () {
286
340
  return earliestDeadline(earliestDeadline(yield* publication.pendingDeadline, yield* messages.pendingDeadline), earliestDeadline(yield* projectionDeadline, yield* host.pendingDeadline));
@@ -288,6 +342,65 @@ var ThreadMaintenance = class ThreadMaintenance extends Context.Service()("@effe
288
342
  const maintenancePassGate = yield* Semaphore.make(1);
289
343
  const minimumAlarmDelay = Math.max(1, Math.ceil(config.alarmBackoffBase / 2));
290
344
  const runTransaction = yield* makeStorageOperation;
345
+ const recoveryStatus = Effect.fn("ThreadMaintenance.recoveryStatus")(function* (threadId) {
346
+ yield* authorizer.authorize(OperationAuthorizationRequest.make({
347
+ operation: "explain",
348
+ threadId
349
+ }));
350
+ return yield* runTransaction("read Thread recovery status", async () => {
351
+ const encoded = await ctx.storage.get(recoveryFaultKey(threadId));
352
+ return encoded === void 0 ? Option.none() : Option.some(decodeRecoveryFault(threadId, encoded));
353
+ });
354
+ });
355
+ const recordRecoveryStatus = Effect.fn("ThreadMaintenance.recordRecoveryStatus")(function* (result) {
356
+ const threads = /* @__PURE__ */ new Map();
357
+ for (const report of result.reports) threads.set(report.threadId, void 0);
358
+ for (const blocked of result.blocked) threads.set(blocked.threadId, blocked.failure);
359
+ if (threads.size === 0) return /* @__PURE__ */ new Map();
360
+ const now = yield* Clock.currentTimeMillis;
361
+ yield* failpoint.hit("maintenance:recovery-status:before");
362
+ const retained = yield* runTransaction("record Thread recovery status", () => ctx.storage.transaction(async (transaction) => {
363
+ const newlyBlocked = [];
364
+ const faults = /* @__PURE__ */ new Map();
365
+ for (const [threadId, failure] of threads) {
366
+ const key = recoveryFaultKey(threadId);
367
+ const encoded = await transaction.get(key);
368
+ const previous = encoded === void 0 ? void 0 : decodeRecoveryFault(threadId, encoded);
369
+ if (failure === void 0) {
370
+ if (previous !== void 0) await transaction.delete(key);
371
+ continue;
372
+ }
373
+ const fault = ThreadRecoveryFault.make({
374
+ schemaVersion: 1,
375
+ threadId,
376
+ firstFailedAt: previous?.firstFailedAt ?? now,
377
+ lastFailedAt: now,
378
+ attempts: Math.min(2147483647, (previous?.attempts ?? 0) + 1),
379
+ retryAt: now + Math.min(6e4, 5e3 * 2 ** Math.min(30, previous?.attempts ?? 0)),
380
+ failure
381
+ });
382
+ await transaction.put(key, encodeRecoveryFault(fault));
383
+ faults.set(threadId, fault);
384
+ if (previous === void 0) newlyBlocked.push(fault);
385
+ }
386
+ return {
387
+ newlyBlocked,
388
+ faults
389
+ };
390
+ }));
391
+ yield* failpoint.hit("maintenance:recovery-status:after");
392
+ for (const fault of retained.newlyBlocked) yield* Effect.logError("Native Thread recovery blocked; accepted work remains pending", fault.failure.reason === "defect" ? Cause.die(fault.failure) : Cause.fail(fault.failure)).pipe(Effect.annotateLogs({ threadId: fault.threadId }));
393
+ return retained.faults;
394
+ });
395
+ const recoverThread = Effect.fn("ThreadMaintenance.recoverThread")(function* (threadId, recovery) {
396
+ const result = yield* runtime.runRecovery({ threadId });
397
+ const faults = yield* recordRecoveryStatus(result);
398
+ for (const report of result.reports) recovery.reports.set(report.submissionId, report);
399
+ recovery.faults.delete(threadId);
400
+ for (const [id, fault] of faults) recovery.faults.set(id, fault);
401
+ recovery.recovered += result.reports.length + result.blocked.length;
402
+ recovery.repaired ||= result.reports.some((report) => report.disposition === "repaired");
403
+ });
291
404
  const ensureAlarm = Effect.fn("ThreadMaintenance.ensureAlarm")(function* () {
292
405
  yield* failpoint.hit("maintenance:ensure:before");
293
406
  const now = yield* Clock.currentTimeMillis;
@@ -368,7 +481,7 @@ var ThreadMaintenance = class ThreadMaintenance extends Context.Service()("@effe
368
481
  };
369
482
  }));
370
483
  });
371
- const advance = Effect.fn("ThreadMaintenance.advance")(function* (started, yieldAfter, observed) {
484
+ const advance = Effect.fn("ThreadMaintenance.advance")(function* (started, yieldAfter, observed, recovery, dispatch = true) {
372
485
  const deadline = yield* publication.pendingDeadline;
373
486
  if (started._tag === "Actionable" || Option.isSome(deadline) && deadline.value <= (yield* Clock.currentTimeMillis)) yield* publication.drain;
374
487
  const pending = yield* publication.pendingDeadline;
@@ -382,43 +495,91 @@ var ThreadMaintenance = class ThreadMaintenance extends Context.Service()("@effe
382
495
  }));
383
496
  return {
384
497
  phase: "caught-up",
385
- recovered: 0,
386
498
  settled: 0,
387
499
  nonterminal: started.nonterminal,
388
500
  nextAttemptAt
389
501
  };
390
502
  }
391
503
  observed.nativeOnly = true;
392
- const recovered = yield* runtime.runRecovery;
393
- const reports = new Map(recovered.map((report) => [report.submissionId, report]));
504
+ const observation = recovery.observation ??= {
505
+ generation: started.generation,
506
+ activeAtStart: started.activeAtStart
507
+ };
394
508
  const current = yield* Stream.runCollect(ledger.scanNonterminal);
509
+ const selectionTime = yield* Clock.currentTimeMillis;
510
+ yield* runTransaction("read Thread recovery deadlines", async () => {
511
+ for (const threadId of new Set(current.map((row) => row.threadId))) {
512
+ if (recovery.loaded.has(threadId)) continue;
513
+ const encoded = await ctx.storage.get(recoveryFaultKey(threadId));
514
+ if (encoded !== void 0) recovery.faults.set(threadId, decodeRecoveryFault(threadId, encoded));
515
+ recovery.loaded.add(threadId);
516
+ }
517
+ });
518
+ const reports = recovery.reports;
519
+ const recoveryFaults = recovery.faults;
520
+ const waiting = (row) => !recovery.pending.has(row.threadId) && !recoveryFaults.has(row.threadId) && stableExternalWait(row, reports);
395
521
  const heads = /* @__PURE__ */ new Map();
396
522
  for (const row of current) {
397
- if (row.state === "unknown" && stableExternalWait(row, reports)) continue;
523
+ if (row.state === "unknown" && waiting(row)) continue;
398
524
  if (!heads.has(row.threadId)) heads.set(row.threadId, row);
399
525
  }
400
- const eligible = [...heads.values()].filter((head) => !stableExternalWait(head, reports)).map((head) => head.threadId).sort();
401
- const selectionTime = yield* Clock.currentTimeMillis;
526
+ const stopping = /* @__PURE__ */ new Set();
527
+ for (const head of heads.values()) {
528
+ if (head.state !== "ready" || recovery.pending.has(head.threadId) || recoveryFaults.has(head.threadId)) continue;
529
+ const intent = yield* isolateRecovery(ledger.readAbortIntent(AbortIntentRequest.make({ submissionId: head.submissionId })), {
530
+ timeout: recoveryConfig.recoveryTimeout,
531
+ phase: () => "recovery",
532
+ operation: "read abort intent"
533
+ });
534
+ if (Result.isSuccess(intent)) {
535
+ if (intent.success !== void 0) stopping.add(head.threadId);
536
+ } else {
537
+ const faults = yield* recordRecoveryStatus(RecoverySweepResult.make({
538
+ reports: [],
539
+ blocked: [RecoveryBlocked.make({
540
+ threadId: head.threadId,
541
+ failure: intent.failure
542
+ })]
543
+ }));
544
+ for (const [threadId, fault] of faults) recoveryFaults.set(threadId, fault);
545
+ recovery.recovered++;
546
+ }
547
+ }
548
+ const eligible = [...heads.values()].filter((head) => !recovery.pending.has(head.threadId) && !stopping.has(head.threadId) && !recoveryFaults.has(head.threadId) && !waiting(head) && (head.state === "ready" || reports.has(head.submissionId))).map((head) => head.threadId).sort();
402
549
  yield* failpoint.hit("maintenance:select:before");
403
550
  const selection = yield* runTransaction("select maintenance lane", () => ctx.storage.transaction(async (transaction) => {
404
551
  const { state } = await readMaintenanceState(transaction);
405
552
  const retries = (state.bindingRetries ?? []).filter((retry) => heads.get(retry.threadId)?.submissionId === retry.submissionId);
406
553
  const runnable = eligible.filter((threadId) => !retries.some((retry) => retry.threadId === threadId && retry.notBefore > selectionTime));
407
- const next = runnable.find((threadId) => state.lastServedThreadId === void 0 || threadId > state.lastServedThreadId) ?? runnable[0];
554
+ const next = dispatch ? runnable.find((threadId) => state.lastServedThreadId === void 0 || threadId > state.lastServedThreadId) ?? runnable[0] : void 0;
408
555
  if (next !== void 0) await transaction.put(MAINTENANCE_STATE_KEY, encodeMaintenanceState(ThreadMaintenanceState.make({
409
556
  ...state,
410
557
  lastServedThreadId: next
411
558
  })));
559
+ const backlog = recovery.started ? [] : [...heads.keys()].filter((threadId) => {
560
+ const fault = recoveryFaults.get(threadId);
561
+ return !eligible.includes(threadId) && (fault === void 0 || fault.retryAt <= selectionTime);
562
+ });
563
+ const after = backlog.filter((threadId) => state.lastRecoveredThreadId === void 0 || threadId > state.lastRecoveredThreadId);
564
+ const before = backlog.filter((threadId) => state.lastRecoveredThreadId !== void 0 && threadId <= state.lastRecoveredThreadId);
412
565
  return {
413
566
  selected: next,
414
- retries
567
+ retries,
568
+ backlog: [...after, ...before]
415
569
  };
416
570
  }));
417
571
  yield* failpoint.hit("maintenance:select:after");
572
+ if (!recovery.started) {
573
+ recovery.started = true;
574
+ recovery.needsCheckpoint = selection.backlog.length > 0;
575
+ for (const threadId of selection.backlog) recovery.pending.add(threadId);
576
+ yield* Deferred.succeed(recovery.queue, selection.backlog);
577
+ }
418
578
  const selected = selection.selected === void 0 ? void 0 : heads.get(selection.selected);
579
+ if (selected !== void 0) yield* recoverThread(selected.threadId, recovery);
419
580
  let retries = selection.retries;
420
581
  let bindingFailure;
421
- const settlement = selected === void 0 ? Option.none() : yield* runtime.processThreadHead(selected.threadId, { yieldAfter }).pipe(Effect.catchTag("BindingUnavailable", (failure) => {
582
+ const settlement = selected === void 0 || recoveryFaults.has(selected.threadId) ? Option.none() : yield* runtime.processThreadHead(selected.threadId, { yieldAfter }).pipe(Effect.catchTag("BindingUnavailable", (failure) => {
422
583
  bindingFailure = failure;
423
584
  return Effect.succeed(Option.none());
424
585
  }));
@@ -456,22 +617,22 @@ var ThreadMaintenance = class ThreadMaintenance extends Context.Service()("@effe
456
617
  const remaining = yield* Stream.runCollect(ledger.scanNonterminal);
457
618
  const waitingHeads = /* @__PURE__ */ new Map();
458
619
  const autonomous = remaining.some((snapshot) => {
459
- if (snapshot.state === "unknown" && stableExternalWait(snapshot, reports)) return false;
620
+ if (snapshot.state === "unknown" && waiting(snapshot)) return false;
460
621
  const headWaiting = waitingHeads.get(snapshot.threadId);
461
- if (headWaiting === void 0) waitingHeads.set(snapshot.threadId, stableExternalWait(snapshot, reports));
622
+ if (headWaiting === void 0) waitingHeads.set(snapshot.threadId, waiting(snapshot));
462
623
  if (headWaiting === true && snapshot.state === "ready" && reports.get(snapshot.submissionId)?.decision._tag === "ApplyInput") return false;
463
- return !stableExternalWait(snapshot, reports);
624
+ return !waiting(snapshot);
464
625
  });
465
- const progressed = Option.isSome(settlement) || recovered.some((report) => report.disposition === "repaired");
626
+ const progressed = Option.isSome(settlement) || recovery.repaired;
466
627
  const now = yield* Clock.currentTimeMillis;
467
628
  const ordinaryDelay = autonomous ? yield* rearmDelay(progressed, started.stalls) : 0;
468
- const nextEligible = eligible.map((threadId) => retries.find((retry) => retry.submissionId === heads.get(threadId)?.submissionId)?.notBefore ?? now);
469
- const bindingDelay = nextEligible.length === 0 ? 0 : Math.max(0, Math.min(...nextEligible) - now);
470
- const delay = Math.max(ordinaryDelay, bindingDelay);
629
+ const nextEligible = [...recoveryFaults.values()].map((fault) => fault.retryAt).concat(eligible.map((threadId) => retries.find((retry) => retry.submissionId === heads.get(threadId)?.submissionId)?.notBefore ?? now));
630
+ const retryDelay = nextEligible.length === 0 ? 0 : Math.max(0, Math.min(...nextEligible) - now);
631
+ const delay = Math.max(ordinaryDelay, retryDelay);
471
632
  yield* failpoint.hit("maintenance:checkpoint:before");
472
633
  const nextAttemptAt = yield* mutations.withSnapshot((active) => runTransaction("checkpoint native maintenance", () => ctx.storage.transaction(async (transaction) => {
473
634
  const { state } = await readMaintenanceState(transaction);
474
- const processed = autonomous || started.activeAtStart > 0 || active > 0 ? state.processed : state.processed > started.generation ? state.processed : started.generation;
635
+ const processed = autonomous || observation.activeAtStart > 0 || started.activeAtStart > 0 || active > 0 ? state.processed : state.processed > observation.generation ? state.processed : observation.generation;
475
636
  const next = ThreadMaintenanceState.make({
476
637
  ...Struct.omit(state, ["retry"]),
477
638
  processed,
@@ -491,7 +652,6 @@ var ThreadMaintenance = class ThreadMaintenance extends Context.Service()("@effe
491
652
  yield* failpoint.hit("maintenance:checkpoint:after");
492
653
  return {
493
654
  phase: "actionable",
494
- recovered: recovered.length,
495
655
  settled: Option.isSome(settlement) ? 1 : 0,
496
656
  nonterminal: remaining.length,
497
657
  nextAttemptAt
@@ -499,49 +659,164 @@ var ThreadMaintenance = class ThreadMaintenance extends Context.Service()("@effe
499
659
  });
500
660
  const pass = Effect.fn("ThreadMaintenance.pass")(function* (yieldAfter, dispatchUntil, observed) {
501
661
  const notified = (yield* Stream.toPull(wakes.wakes)).pipe(Effect.asVoid, Effect.catch(() => Effect.never));
502
- let started = yield* beginNative(observed);
662
+ const started = yield* beginNative(observed);
503
663
  const auxiliaryScope = yield* Effect.acquireRelease(Scope.make("parallel"), (scope, exit) => Scope.close(scope, exit));
504
664
  const dispatchClosed = yield* Deferred.make();
505
665
  const stopDispatch = Deferred.await(dispatchClosed);
506
- const deliveryFiber = yield* Effect.forkIn(Scope.provide(auxiliaryScope)(messages.drainUntil(stopDispatch, dispatchUntil)), auxiliaryScope);
666
+ const checks = yield* PubSub.sliding(1);
667
+ yield* Effect.addFinalizer(() => PubSub.shutdown(checks));
668
+ let dispatchOpen = true;
669
+ let activityChanged = yield* Deferred.make();
670
+ const signalActivity = Effect.suspend(() => Deferred.succeed(activityChanged, void 0));
671
+ const closeDispatch = Effect.sync(() => {
672
+ dispatchOpen = false;
673
+ }).pipe(Effect.andThen(Deferred.succeed(dispatchClosed, void 0)));
674
+ const makeActivity = Effect.fnUntraced(function* (allowance) {
675
+ const initialized = yield* Deferred.make();
676
+ let active = 0;
677
+ let failed = false;
678
+ return {
679
+ activity: {
680
+ ready: Deferred.succeed(initialized, void 0).pipe(Effect.andThen(signalActivity), Effect.asVoid, Effect.uninterruptible),
681
+ subscribeChanges: PubSub.subscribe(checks).pipe(Effect.map(PubSub.take)),
682
+ run: (wave) => Effect.acquireUseRelease(Effect.sync(() => {
683
+ if (!dispatchOpen) return false;
684
+ active++;
685
+ return true;
686
+ }), (admitted) => !admitted ? Effect.void : (allowance === void 0 ? wave : wave.pipe(Effect.timeoutOrElse({
687
+ duration: allowance,
688
+ orElse: () => DurableAlarmError.make({
689
+ operation: "host dispatch allowance",
690
+ message: "The admitted host wave exceeded its allowance; durable work remains pending"
691
+ })
692
+ }))).pipe(Effect.onExit((exit) => Effect.sync(() => {
693
+ if (Exit.isFailure(exit)) failed = true;
694
+ }))), (admitted) => Effect.sync(() => {
695
+ if (admitted) active--;
696
+ }).pipe(Effect.andThen(signalActivity)))
697
+ },
698
+ initialized: Deferred.await(initialized),
699
+ quiet: () => active === 0 && Deferred.isDoneUnsafe(initialized),
700
+ failed: () => failed
701
+ };
702
+ });
703
+ const messageActivity = yield* makeActivity();
704
+ const hostActivity = yield* makeActivity(host.dispatchTimeoutMillis);
705
+ const recovery = {
706
+ queue: yield* Deferred.make(),
707
+ pending: /* @__PURE__ */ new Set(),
708
+ loaded: /* @__PURE__ */ new Set(),
709
+ reports: /* @__PURE__ */ new Map(),
710
+ faults: /* @__PURE__ */ new Map(),
711
+ started: false,
712
+ needsCheckpoint: false,
713
+ recovered: 0,
714
+ repaired: false
715
+ };
716
+ const recoveryFiber = yield* Effect.forkIn(Effect.gen(function* () {
717
+ for (const threadId of yield* Deferred.await(recovery.queue)) {
718
+ yield* failpoint.hit("maintenance:select:before");
719
+ yield* runTransaction("select old recovery lane", () => ctx.storage.transaction(async (transaction) => {
720
+ const { state } = await readMaintenanceState(transaction);
721
+ await transaction.put(MAINTENANCE_STATE_KEY, encodeMaintenanceState(ThreadMaintenanceState.make({
722
+ ...state,
723
+ lastRecoveredThreadId: threadId
724
+ })));
725
+ }));
726
+ yield* failpoint.hit("maintenance:select:after");
727
+ yield* recoverThread(threadId, recovery);
728
+ recovery.pending.delete(threadId);
729
+ yield* wakes.notify(threadId);
730
+ }
731
+ }), auxiliaryScope);
732
+ const deliveryFiber = yield* Effect.forkIn(Scope.provide(auxiliaryScope)(messages.drainUntil(stopDispatch, dispatchUntil).pipe(Effect.provideService(ThreadMaintenanceActivity, messageActivity.activity), Effect.onExit(() => messageActivity.activity.ready))), auxiliaryScope);
507
733
  const hostFiber = yield* Effect.forkIn(Scope.provide(auxiliaryScope)(Effect.gen(function* () {
508
734
  yield* Schema.decodeEffect(AuxiliaryDispatchMillis)(host.dispatchTimeoutMillis).pipe(Effect.mapError((cause) => DurableAlarmError.make({
509
735
  operation: "host dispatch allowance",
510
736
  message: "Declare an integer whole-wave allowance between 1 and 300000 milliseconds",
511
737
  cause
512
738
  })));
513
- yield* host.drainUntil(stopDispatch, dispatchUntil);
514
- })), auxiliaryScope);
515
- const backfill = yield* Effect.forkIn(drainDue.pipe(Effect.provideService(ThreadProjectionMaintenance, projection), Effect.timeoutOption(config.projectionDispatchTimeoutMillis)), auxiliaryScope);
516
- let result = yield* advance(started, yieldAfter, observed);
739
+ const initialization = hostActivity.initialized.pipe(Effect.timeoutOrElse({
740
+ duration: host.dispatchTimeoutMillis,
741
+ orElse: () => DurableAlarmError.make({
742
+ operation: "host initialization",
743
+ message: "The host did not account for initial maintenance before its allowance; durable work remains pending"
744
+ })
745
+ }), Effect.andThen(Effect.never));
746
+ yield* Effect.raceFirst(host.drainUntil(stopDispatch, dispatchUntil).pipe(Effect.provideService(ThreadMaintenanceActivity, hostActivity.activity)), initialization);
747
+ }).pipe(Effect.onExit(() => hostActivity.activity.ready))), auxiliaryScope);
748
+ const backfill = yield* Effect.forkIn(drainDue.pipe(Effect.provideService(ThreadProjectionMaintenance, projection), Effect.timeoutOption(config.projectionDispatchTimeoutMillis), Effect.onExit(() => signalActivity)), auxiliaryScope);
749
+ let result = yield* advance(started, yieldAfter, observed, recovery);
750
+ recovery.started = true;
751
+ yield* Deferred.succeed(recovery.queue, []);
517
752
  let phase = result.phase;
518
- let recovered = result.recovered;
519
753
  let settled = result.settled;
520
754
  observed.nativeOnly = false;
521
- yield* Deferred.succeed(dispatchClosed, void 0);
522
- const remaining = Math.max(1, DateTime.toEpochMillis(dispatchUntil) - (yield* Clock.currentTimeMillis));
523
- const hostJoin = yield* Effect.forkIn(Fiber.join(hostFiber).pipe(Effect.timeoutOption(Math.min(host.dispatchTimeoutMillis, remaining)), Effect.tap((outcome) => Effect.annotateCurrentSpan({ "host.timedOut": Option.isNone(outcome) }))), auxiliaryScope);
755
+ const hostJoin = yield* Effect.forkIn(Effect.gen(function* () {
756
+ yield* stopDispatch;
757
+ const remaining = Math.max(1, DateTime.toEpochMillis(dispatchUntil) - (yield* Clock.currentTimeMillis));
758
+ const outcome = yield* Fiber.join(hostFiber).pipe(Effect.timeoutOption(Math.min(host.dispatchTimeoutMillis, remaining)));
759
+ yield* Effect.annotateCurrentSpan({ "host.timedOut": Option.isNone(outcome) });
760
+ }), auxiliaryScope);
524
761
  const retired = yield* Effect.forkChild(Fiber.joinAll([
525
762
  deliveryFiber,
526
763
  hostJoin,
527
- backfill
764
+ backfill,
765
+ recoveryFiber
528
766
  ]));
529
- const auxiliaryPending = () => deliveryFiber.pollUnsafe() === void 0 || hostFiber.pollUnsafe() === void 0 || backfill.pollUnsafe() === void 0;
530
- while (retired.pollUnsafe() === void 0 && auxiliaryPending()) {
767
+ while (true) {
768
+ const recoveryFinished = recoveryFiber.pollUnsafe() !== void 0;
769
+ const retiredExit = retired.pollUnsafe();
770
+ if (recoveryFinished) {
771
+ if (recovery.needsCheckpoint) {
772
+ yield* Fiber.join(recoveryFiber);
773
+ result = yield* advance(started, yieldAfter, observed, recovery, settled === 0);
774
+ if (result.phase === "actionable") phase = "actionable";
775
+ settled += result.settled;
776
+ recovery.needsCheckpoint = false;
777
+ yield* PubSub.publish(checks, void 0);
778
+ }
779
+ }
780
+ if (retiredExit !== void 0 && Exit.isFailure(retiredExit)) break;
781
+ if (recoveryFinished && backfill.pollUnsafe() !== void 0 && messageActivity.quiet() && hostActivity.quiet()) {
782
+ const now = yield* Clock.currentTimeMillis;
783
+ const messageDeadline = yield* messages.pendingDeadline;
784
+ const hostDeadline = yield* host.pendingDeadline;
785
+ const messageDue = deliveryFiber.pollUnsafe() === void 0 && !messageActivity.failed() && Option.isSome(messageDeadline) && messageDeadline.value <= now;
786
+ const hostDue = hostFiber.pollUnsafe() === void 0 && !hostActivity.failed() && now + host.dispatchTimeoutMillis <= DateTime.toEpochMillis(dispatchUntil) && Option.isSome(hostDeadline) && hostDeadline.value <= now;
787
+ if (yield* Effect.sync(() => {
788
+ if (messageDue || hostDue || !messageActivity.quiet() || !hostActivity.quiet()) return false;
789
+ dispatchOpen = false;
790
+ return true;
791
+ })) break;
792
+ yield* PubSub.publish(checks, void 0);
793
+ }
531
794
  const now = yield* Clock.currentTimeMillis;
532
795
  const until = DateTime.toEpochMillis(yieldAfter);
533
796
  if (now >= until) break;
534
797
  const next = Math.min(result.nextAttemptAt ?? Infinity, now + config.wakeScanInterval, until);
535
- if (!(yield* Effect.raceFirst(Effect.raceFirst(notified, Effect.sleep(Math.max(0, next - now))).pipe(Effect.as(true)), Fiber.await(retired).pipe(Effect.as(false)))) || retired.pollUnsafe() !== void 0 || !auxiliaryPending()) break;
798
+ const recoveryDone = recoveryFinished ? Effect.never : Fiber.await(recoveryFiber);
799
+ const retirementDone = retiredExit === void 0 ? Fiber.await(retired) : Effect.never;
800
+ const changed = activityChanged;
801
+ const ready = yield* Effect.raceFirst(Effect.raceFirst(notified, Effect.sleep(Math.max(0, next - now))).pipe(Effect.as("native")), Effect.raceFirst(recoveryDone.pipe(Effect.as("recovery")), Effect.raceFirst(retirementDone.pipe(Effect.as("retired")), Deferred.await(changed).pipe(Effect.as("activity")))));
802
+ if (ready === "recovery" || ready === "retired") continue;
803
+ if (ready === "activity") activityChanged = yield* Deferred.make();
536
804
  if ((yield* Clock.currentTimeMillis) >= until) break;
537
- started = yield* beginNative(observed);
538
- result = yield* advance(started, yieldAfter, observed);
805
+ if (ready === "native") yield* PubSub.publish(checks, void 0);
806
+ const awakened = yield* beginNative(observed);
807
+ result = yield* advance(awakened, yieldAfter, observed, recovery);
539
808
  if (result.phase === "actionable") phase = "actionable";
540
- recovered += result.recovered;
541
809
  settled += result.settled;
542
810
  observed.nativeOnly = false;
811
+ if (result.settled > 0) yield* PubSub.publish(checks, void 0);
543
812
  }
813
+ yield* closeDispatch;
544
814
  yield* Fiber.join(retired);
815
+ if (recovery.needsCheckpoint) {
816
+ result = yield* advance(started, yieldAfter, observed, recovery, false);
817
+ if (result.phase === "actionable") phase = "actionable";
818
+ settled += result.settled;
819
+ }
545
820
  yield* Scope.close(auxiliaryScope, Exit.void);
546
821
  yield* failpoint.hit("maintenance:finish:before");
547
822
  const disposition = yield* mutations.withSnapshot((active) => Effect.gen(function* () {
@@ -562,7 +837,7 @@ var ThreadMaintenance = class ThreadMaintenance extends Context.Service()("@effe
562
837
  yield* failpoint.hit("maintenance:finish:after");
563
838
  const report = MaintenancePassReport.make({
564
839
  phase,
565
- recovered,
840
+ recovered: recovery.recovered,
566
841
  settled,
567
842
  nonterminal: result.nonterminal,
568
843
  alarm: disposition
@@ -591,11 +866,12 @@ var ThreadMaintenance = class ThreadMaintenance extends Context.Service()("@effe
591
866
  })
592
867
  })),
593
868
  ensureAlarm: mutations.withSnapshot(() => ensureAlarm()),
869
+ recoveryStatus,
594
870
  withMutation: (body) => mutations.withMutation(body.pipe(Effect.tap(() => publishCommitted.pipe(Effect.provideService(ThreadPublication, publication)))))
595
871
  });
596
872
  }));
597
873
  };
598
874
  //#endregion
599
- export { DurableAlarmError, DurableAlarmService, MaintenancePassReport, ThreadHostMaintenance, ThreadMaintenance, ThreadMaintenanceFailpoint, ThreadMessageDelivery, ThreadMutationGate, ThreadPublication, publishCommitted, Alarm_exports as t };
875
+ export { DurableAlarmError, DurableAlarmService, MaintenancePassReport, ThreadHostMaintenance, ThreadMaintenance, ThreadMaintenanceActivity, ThreadMaintenanceFailpoint, ThreadMessageDelivery, ThreadMutationGate, ThreadPublication, ThreadRecoveryFault, publishCommitted, Alarm_exports as t };
600
876
 
601
877
  //# sourceMappingURL=Alarm.mjs.map