@effect-agent/platform-cloudflare 0.1.0-beta.111 → 0.1.0-beta.113

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/Alarm.ts CHANGED
@@ -9,8 +9,10 @@ import {
9
9
  Fiber,
10
10
  Layer,
11
11
  Option,
12
+ PubSub,
12
13
  Random,
13
14
  Ref,
15
+ Result,
14
16
  Schema,
15
17
  Scope,
16
18
  Semaphore,
@@ -20,11 +22,25 @@ import {
20
22
  import { type DurableBindingFailure } from "effect-agent/agent-registration";
21
23
  import {
22
24
  DurableAgentRuntime,
25
+ DurableRuntimeConfig,
26
+ isolateRecovery,
27
+ RecoveryBlocked,
28
+ RecoveryFailure,
23
29
  type DurableWorkerFailure,
24
30
  type RecoveryReport,
31
+ RecoverySweepResult,
25
32
  } from "effect-agent/durable-agent-runtime";
26
33
  import { ThreadId, SubmissionId } from "effect-agent/identifiers";
27
- import { SubmissionLedger, type SubmissionSnapshot } from "effect-agent/submission-ledger";
34
+ import {
35
+ OperationAuthorizationRequest,
36
+ OperationAuthorizer,
37
+ type OperationDenied,
38
+ } from "effect-agent/operation-authorizer";
39
+ import {
40
+ AbortIntentRequest,
41
+ SubmissionLedger,
42
+ type SubmissionWorkItem,
43
+ } from "effect-agent/submission-ledger";
28
44
  import {
29
45
  ThreadProjectionMaintenance,
30
46
  drainDue,
@@ -187,7 +203,7 @@ export class MaintenancePassReport extends Schema.Class<MaintenancePassReport>(
187
203
  )({
188
204
  /** `caught-up` ran no runtime work (publication may be pending); `actionable` ran recovery. */
189
205
  phase: Schema.Literals(["caught-up", "actionable"]),
190
- /** Recovery decisions executed (or deferred) BEFORE any new claim in this pass. */
206
+ /** Recovery decisions and Thread faults observed during this event. */
191
207
  recovered: Schema.Int.check(Schema.isGreaterThanOrEqualTo(0)),
192
208
  /** Head Attempts settled during the event. Joined input may settle with each head. */
193
209
  settled: Schema.Int.check(Schema.isGreaterThanOrEqualTo(0)),
@@ -211,6 +227,8 @@ export type ThreadMaintenanceFailpointLocation =
211
227
  | "maintenance:select:after"
212
228
  | "maintenance:binding-retry:before"
213
229
  | "maintenance:binding-retry:after"
230
+ | "maintenance:recovery-status:before"
231
+ | "maintenance:recovery-status:after"
214
232
  | "maintenance:retry:before"
215
233
  | "maintenance:retry:after"
216
234
  | "maintenance:checkpoint:before"
@@ -268,6 +286,65 @@ export class ThreadPublication extends Context.Service<
268
286
  });
269
287
  }
270
288
 
289
+ /** Event-local accounting shared by the existing maintenance pumps and native scheduler. */
290
+ export class ThreadMaintenanceActivity extends Context.Service<
291
+ ThreadMaintenanceActivity,
292
+ {
293
+ /**
294
+ * Account for one finite selection/dispatch, through joined completion or interruption.
295
+ * Register before reading/claiming work. After dispatch closes the body is not started.
296
+ * Host waves retain their declared allowance from this call; message Claims keep the driver's
297
+ * own deadline. Waiting for notifications belongs outside this bracket.
298
+ */
299
+ readonly run: <R>(
300
+ wave: Effect.Effect<void, DurableAlarmError, R>,
301
+ ) => Effect.Effect<void, DurableAlarmError, R>;
302
+ /** Acknowledge all initial subscriptions/scans. Composite hosts use `all` below. */
303
+ readonly ready: Effect.Effect<void>;
304
+ /**
305
+ * Acquire an independent scoped subscription before the initial scan and `ready`.
306
+ * The returned wait observes coalesced native scan hints, including those published before
307
+ * its first execution. Each hint requests a current local due-work check.
308
+ */
309
+ readonly subscribeChanges: Effect.Effect<Effect.Effect<void>, never, Scope.Scope>;
310
+ }
311
+ >()("@effect-agent/platform-cloudflare/ThreadMaintenanceActivity") {
312
+ /** Join child pumps; acknowledge the parent only after every child is ready or has exited. */
313
+ static readonly all = Effect.fnUntraced(function* <A, E, R>(
314
+ lanes: ReadonlyArray<Effect.Effect<A, E, R>>,
315
+ ): Effect.fn.Return<ReadonlyArray<A>, E, R | ThreadMaintenanceActivity> {
316
+ const activity = yield* ThreadMaintenanceActivity;
317
+ let remaining = lanes.length;
318
+
319
+ if (remaining === 0) return yield* activity.ready.pipe(Effect.as([]));
320
+
321
+ return yield* Effect.forEach(
322
+ lanes,
323
+ (lane) => {
324
+ let ready = false;
325
+
326
+ const child: ThreadMaintenanceActivity["Service"] = {
327
+ ...activity,
328
+ ready: Effect.suspend(() => {
329
+ if (ready) return Effect.void;
330
+ ready = true;
331
+
332
+ return --remaining === 0 ? activity.ready : Effect.void;
333
+ }).pipe(Effect.uninterruptible),
334
+ };
335
+
336
+ // A failed initial setup is accounted for without swallowing its Cause. Callers
337
+ // may collect Exits when sibling pumps must keep their independent opportunities.
338
+ return lane.pipe(
339
+ Effect.provideService(ThreadMaintenanceActivity, child),
340
+ Effect.onExit(() => child.ready),
341
+ );
342
+ },
343
+ { concurrency: "unbounded" },
344
+ );
345
+ });
346
+ }
347
+
271
348
  /**
272
349
  * Host-assembled native message recovery. The driver bounds each actual Claim and persists its
273
350
  * timeout/retry before this pump returns. Do not add a second timer starting at batch selection:
@@ -277,7 +354,7 @@ export const ThreadMessageDelivery = Context.Reference<{
277
354
  readonly drainUntil: (
278
355
  dispatchClosed: Effect.Effect<void>,
279
356
  dispatchUntil: DateTime.Utc,
280
- ) => Effect.Effect<void, DurableAlarmError, Scope.Scope>;
357
+ ) => Effect.Effect<void, DurableAlarmError, Scope.Scope | ThreadMaintenanceActivity>;
281
358
  readonly pendingDeadline: Effect.Effect<Option.Option<number>, DurableAlarmError>;
282
359
  }>("@effect-agent/platform-cloudflare/ThreadMessageDelivery", {
283
360
  defaultValue: () => ({
@@ -288,14 +365,16 @@ export const ThreadMessageDelivery = Context.Reference<{
288
365
 
289
366
  /**
290
367
  * Application obligations sharing this Object's alarm. Admit one initial external wave even on
291
- * a caught-up pass, then respond to wakes until dispatchClosed. This closes only admission
292
- * of NEW external waves; native work can continue while admitted waves finish. Keep local
293
- * admission/hub subscriptions in the event Scope until maintenance tears it down.
368
+ * a caught-up pass, then respond to wakes and ThreadMaintenanceActivity.subscribeChanges until
369
+ * dispatchClosed. Yield the event's ThreadMaintenanceActivity to bracket each finite lane with
370
+ * run and acknowledge initial setup with ready.
371
+ * Passive subscriptions do not count as active work. Closure stops both new waves and native
372
+ * Attempts; local admission/control subscriptions belong to the event Scope until teardown.
294
373
  * No deadline sleeps or automatic retry loops. Return after already-admitted waves finish.
295
374
  *
296
375
  * Declare a finite whole-wave allowance (1..300000ms): maximum for parallel lanes, sum for
297
376
  * sequential operations. Admit a wave only if its allowance fits before dispatchUntil. Later
298
- * arrivals cannot renew the retirement window. Maintenance bounds the join after dispatchClosed,
377
+ * arrivals cannot renew an admitted wave's allowance. Maintenance bounds each registered wave,
299
378
  * interrupts and joins event Scope, then reads local deadlines under the mutation gate.
300
379
  *
301
380
  * Setup and pendingDeadline are bounded local operations. Persist claims/envelopes before
@@ -309,7 +388,7 @@ export const ThreadHostMaintenance = Context.Reference<{
309
388
  readonly drainUntil: (
310
389
  dispatchClosed: Effect.Effect<void>,
311
390
  dispatchUntil: DateTime.Utc,
312
- ) => Effect.Effect<void, DurableAlarmError, Scope.Scope>;
391
+ ) => Effect.Effect<void, DurableAlarmError, Scope.Scope | ThreadMaintenanceActivity>;
313
392
  readonly pendingDeadline: Effect.Effect<Option.Option<number>, DurableAlarmError>;
314
393
  }>("@effect-agent/platform-cloudflare/ThreadHostMaintenance", {
315
394
  defaultValue: () => ({
@@ -354,14 +433,64 @@ class BindingRetry extends Schema.Class<BindingRetry>("BindingRetry")({
354
433
  reportedAt: Schema.Finite,
355
434
  }) {}
356
435
 
436
+ /**
437
+ * A durable observation of blocked recovery, independent of the execution journal. This is
438
+ * neither a Settlement nor proof that external effects did not happen. A successful recovery
439
+ * sweep clears it; repair must preserve canonical history and the original admission identity.
440
+ */
441
+ export class ThreadRecoveryFault extends Schema.Class<ThreadRecoveryFault>(
442
+ "@effect-agent/platform-cloudflare/ThreadRecoveryFault",
443
+ )({
444
+ schemaVersion: Schema.Literal(1),
445
+ threadId: ThreadId,
446
+ firstFailedAt: Schema.Finite,
447
+ lastFailedAt: Schema.Finite,
448
+ /** Earliest automatic recovery retry; new admissions do not erase this deadline. */
449
+ retryAt: Schema.Finite,
450
+ /** Saturates at 2^31 - 1; one observation per Thread per recovery sweep. */
451
+ attempts: Schema.Int.check(Schema.isGreaterThan(0), Schema.isLessThanOrEqualTo(2_147_483_647)),
452
+ failure: RecoveryFailure,
453
+ }) {}
454
+
455
+ const recoveryFaultKey = (threadId: ThreadId) =>
456
+ `effect-agent:thread-recovery-fault:v1:${threadId}`;
457
+
458
+ const decodeRecoveryFaultValue = Schema.decodeUnknownSync(ThreadRecoveryFault);
459
+
460
+ const decodeRecoveryFault = (threadId: ThreadId, encoded: unknown) => {
461
+ const fault = decodeRecoveryFaultValue(encoded);
462
+
463
+ if (fault.threadId !== threadId) throw new Error("Recovery status does not match its Thread key");
464
+
465
+ return fault;
466
+ };
467
+
468
+ const encodeRecoveryFault = Schema.encodeSync(ThreadRecoveryFault);
469
+
357
470
  interface NativePassResult {
358
471
  readonly phase: "caught-up" | "actionable";
359
- readonly recovered: number;
360
472
  readonly settled: number;
361
473
  readonly nonterminal: number;
362
474
  readonly nextAttemptAt: number | undefined;
363
475
  }
364
476
 
477
+ /** Event-local observations only; durable ingress keeps racing mutations dirty. */
478
+ interface NativeRecovery {
479
+ readonly queue: Deferred.Deferred<ReadonlyArray<ThreadId>>;
480
+ readonly pending: Set<ThreadId>;
481
+ readonly loaded: Set<ThreadId>;
482
+ readonly reports: Map<SubmissionId, RecoveryReport>;
483
+ readonly faults: Map<ThreadId, ThreadRecoveryFault>;
484
+ observation?: {
485
+ readonly generation: bigint;
486
+ readonly activeAtStart: number;
487
+ };
488
+ started: boolean;
489
+ needsCheckpoint: boolean;
490
+ recovered: number;
491
+ repaired: boolean;
492
+ }
493
+
365
494
  interface MaintenanceObservation {
366
495
  generation?: bigint;
367
496
  nativeOnly: boolean;
@@ -384,6 +513,8 @@ class ThreadMaintenanceState extends Schema.Class<ThreadMaintenanceState>(
384
513
  nonterminal: Schema.Int.check(Schema.isGreaterThanOrEqualTo(0)),
385
514
  /** One physical-owner cursor; old single-lane records need no conversion. */
386
515
  lastServedThreadId: Schema.optionalKey(ThreadId),
516
+ /** Rotate old recovery independently of dispatch, including after eviction or timeout. */
517
+ lastRecoveredThreadId: Schema.optionalKey(ThreadId),
387
518
  bindingRetries: Schema.optionalKey(Schema.Array(BindingRetry)),
388
519
  /** Absent on older records. A newer mutation makes this retry obsolete. */
389
520
  retry: Schema.optionalKey(MaintenanceRetry),
@@ -425,7 +556,7 @@ const ensureTransactionAlarmBy = async (
425
556
  };
426
557
 
427
558
  const stableExternalWait = (
428
- snapshot: SubmissionSnapshot,
559
+ snapshot: SubmissionWorkItem,
429
560
  reports: ReadonlyMap<string, RecoveryReport>,
430
561
  ): boolean => {
431
562
  const report = reports.get(snapshot.submissionId);
@@ -553,10 +684,10 @@ export type MaintenancePassFailure =
553
684
  * recovery, ledger scans or canonical-history reads.
554
685
  * 2. Reconcile before each head Attempt, then checkpoint only the observed generation. A racing
555
686
  * producer keeps its newer generation dirty. Native retries retain their durable backoff.
556
- * 3. After the initial native opportunity, stop admitting new external waves and let the
557
- * active waves finish. While they remain in flight, native wakes and bounded scans can
558
- * advance more heads. All Attempts share the event's original ten-minute yield deadline.
559
- * 4. Native message delivery retains its driver-owned Claim deadline. Host/backfill joins are
687
+ * 3. Keep native and delivery admission open together while finite waves remain active, so
688
+ * fresh replies and abort controls can progress during unrelated cleanup. Close atomically
689
+ * at quiescence, or at the original ten-minute yield deadline, before retiring listeners.
690
+ * 4. Native message delivery retains its driver-owned Claim deadline. Host/backfill waves are
560
691
  * bounded independently; incoming native work never restarts or cancels their attempts.
561
692
  * Auxiliary failures are reported after the current native opportunity.
562
693
  * 5. Close every event resource before the final gated deadline snapshot and alarm decision.
@@ -572,6 +703,15 @@ export class ThreadMaintenance extends Context.Service<
572
703
  * generation has an alarm. It never scans the ledger or canonical history.
573
704
  */
574
705
  readonly ensureAlarm: Effect.Effect<void, MaintenancePassFailure>;
706
+ /**
707
+ * Authorize `explain`, then read one bounded local record without reading execution history.
708
+ * None means no recorded fault, not proof of health or settlement. The host authenticates
709
+ * callers and verifies local Thread membership before exposing this service across RPC.
710
+ * Provide OperationAuthorizer when constructing this Layer, as for DurableAgentRuntime.
711
+ */
712
+ readonly recoveryStatus: (
713
+ threadId: ThreadId,
714
+ ) => Effect.Effect<Option.Option<ThreadRecoveryFault>, DurableAlarmError | OperationDenied>;
575
715
  /**
576
716
  * Serialize the pre-arm boundary with pass acknowledgement, advance the durable dirty
577
717
  * generation and arm the alarm in one transaction BEFORE running the caller's mutation.
@@ -589,6 +729,7 @@ export class ThreadMaintenance extends Context.Service<
589
729
  | ThreadPublication
590
730
  | ThreadProjectionMaintenance
591
731
  | DurableAgentRuntime
732
+ | DurableRuntimeConfig
592
733
  | SubmissionLedger
593
734
  | WakeScheduler
594
735
  | DurableAlarmService
@@ -599,6 +740,7 @@ export class ThreadMaintenance extends Context.Service<
599
740
  > = Layer.effect(ThreadMaintenance)(
600
741
  Effect.gen(function* () {
601
742
  const runtime = yield* DurableAgentRuntime;
743
+ const recoveryConfig = yield* DurableRuntimeConfig;
602
744
  const ledger = yield* SubmissionLedger;
603
745
  const wakes = yield* WakeScheduler;
604
746
  const alarm = yield* DurableAlarmService;
@@ -611,6 +753,7 @@ export class ThreadMaintenance extends Context.Service<
611
753
  const projection = yield* ThreadProjectionMaintenance;
612
754
  const messages = yield* ThreadMessageDelivery;
613
755
  const host = yield* ThreadHostMaintenance;
756
+ const authorizer = yield* OperationAuthorizer;
614
757
 
615
758
  // A broken disposable index still needs a retry alarm and must not prevent startup.
616
759
  const projectionDeadline = projection.pendingDeadline.pipe(
@@ -635,6 +778,97 @@ export class ThreadMaintenance extends Context.Service<
635
778
 
636
779
  const runTransaction = yield* makeStorageOperation;
637
780
 
781
+ const recoveryStatus = Effect.fn("ThreadMaintenance.recoveryStatus")(function* (
782
+ threadId: ThreadId,
783
+ ) {
784
+ yield* authorizer.authorize(
785
+ OperationAuthorizationRequest.make({ operation: "explain", threadId }),
786
+ );
787
+
788
+ return yield* runTransaction("read Thread recovery status", async () => {
789
+ const encoded = await ctx.storage.get(recoveryFaultKey(threadId));
790
+
791
+ return encoded === undefined
792
+ ? Option.none()
793
+ : Option.some(decodeRecoveryFault(threadId, encoded));
794
+ });
795
+ });
796
+
797
+ const recordRecoveryStatus = Effect.fn("ThreadMaintenance.recordRecoveryStatus")(function* (
798
+ result: RecoverySweepResult,
799
+ ) {
800
+ const threads = new Map<ThreadId, RecoveryFailure | undefined>();
801
+
802
+ for (const report of result.reports) threads.set(report.threadId, undefined);
803
+ for (const blocked of result.blocked) threads.set(blocked.threadId, blocked.failure);
804
+ if (threads.size === 0) return new Map<ThreadId, ThreadRecoveryFault>();
805
+ const now = yield* Clock.currentTimeMillis;
806
+
807
+ yield* failpoint.hit("maintenance:recovery-status:before");
808
+
809
+ const retained = yield* runTransaction("record Thread recovery status", () =>
810
+ ctx.storage.transaction(async (transaction) => {
811
+ const newlyBlocked: Array<ThreadRecoveryFault> = [];
812
+ const faults = new Map<ThreadId, ThreadRecoveryFault>();
813
+
814
+ for (const [threadId, failure] of threads) {
815
+ const key = recoveryFaultKey(threadId);
816
+ const encoded = await transaction.get(key);
817
+
818
+ const previous =
819
+ encoded === undefined ? undefined : decodeRecoveryFault(threadId, encoded);
820
+
821
+ if (failure === undefined) {
822
+ if (previous !== undefined) await transaction.delete(key);
823
+ continue;
824
+ }
825
+
826
+ const fault = ThreadRecoveryFault.make({
827
+ schemaVersion: 1,
828
+ threadId,
829
+ firstFailedAt: previous?.firstFailedAt ?? now,
830
+ lastFailedAt: now,
831
+ attempts: Math.min(2_147_483_647, (previous?.attempts ?? 0) + 1),
832
+ retryAt: now + Math.min(60_000, 5_000 * 2 ** Math.min(30, previous?.attempts ?? 0)),
833
+ failure,
834
+ });
835
+
836
+ await transaction.put(key, encodeRecoveryFault(fault));
837
+ faults.set(threadId, fault);
838
+ if (previous === undefined) newlyBlocked.push(fault);
839
+ }
840
+
841
+ return { newlyBlocked, faults };
842
+ }),
843
+ );
844
+
845
+ yield* failpoint.hit("maintenance:recovery-status:after");
846
+ for (const fault of retained.newlyBlocked)
847
+ yield* Effect.logError(
848
+ "Native Thread recovery blocked; accepted work remains pending",
849
+ fault.failure.reason === "defect"
850
+ ? Cause.die(fault.failure)
851
+ : Cause.fail(fault.failure),
852
+ ).pipe(Effect.annotateLogs({ threadId: fault.threadId }));
853
+
854
+ return retained.faults;
855
+ });
856
+
857
+ const recoverThread = Effect.fn("ThreadMaintenance.recoverThread")(function* (
858
+ threadId: ThreadId,
859
+ recovery: NativeRecovery,
860
+ ) {
861
+ const result = yield* runtime.runRecovery({ threadId });
862
+ // Visibility is committed before a claim or any fallible auxiliary join.
863
+ const faults = yield* recordRecoveryStatus(result);
864
+
865
+ for (const report of result.reports) recovery.reports.set(report.submissionId, report);
866
+ recovery.faults.delete(threadId);
867
+ for (const [id, fault] of faults) recovery.faults.set(id, fault);
868
+ recovery.recovered += result.reports.length + result.blocked.length;
869
+ recovery.repaired ||= result.reports.some((report) => report.disposition === "repaired");
870
+ });
871
+
638
872
  const ensureAlarm = Effect.fn("ThreadMaintenance.ensureAlarm")(function* () {
639
873
  yield* failpoint.hit("maintenance:ensure:before");
640
874
  const now = yield* Clock.currentTimeMillis;
@@ -821,6 +1055,8 @@ export class ThreadMaintenance extends Context.Service<
821
1055
  started: Effect.Success<ReturnType<typeof beginNative>>,
822
1056
  yieldAfter: DateTime.Utc,
823
1057
  observed: MaintenanceObservation,
1058
+ recovery: NativeRecovery,
1059
+ dispatch = true,
824
1060
  ): Effect.fn.Return<NativePassResult, MaintenancePassFailure> {
825
1061
  const deadline = yield* publication.pendingDeadline;
826
1062
 
@@ -856,33 +1092,99 @@ export class ThreadMaintenance extends Context.Service<
856
1092
 
857
1093
  return {
858
1094
  phase: "caught-up",
859
- recovered: 0,
860
1095
  settled: 0,
861
1096
  nonterminal: started.nonterminal,
862
1097
  nextAttemptAt,
863
1098
  };
864
1099
  }
865
- // Step 2 — reconciliation strictly precedes new work in this pass (exit gate).
1100
+ // Select from control state before reading execution history. A recovering or faulted
1101
+ // Thread cannot enter dispatch; old cleanup has its own scoped opportunity below.
866
1102
  observed.nativeOnly = true;
867
- const recovered: ReadonlyArray<RecoveryReport> = yield* runtime.runRecovery;
868
- const reports = new Map(recovered.map((report) => [report.submissionId, report]));
1103
+
1104
+ // Every checkpoint keeps the producer overlap that belongs to this recovery wave.
1105
+ const observation = (recovery.observation ??= {
1106
+ generation: started.generation,
1107
+ activeAtStart: started.activeAtStart,
1108
+ });
1109
+
869
1110
  const current = yield* Stream.runCollect(ledger.scanNonterminal);
870
- const heads = new Map<ThreadId, SubmissionSnapshot>();
1111
+ const selectionTime = yield* Clock.currentTimeMillis;
1112
+
1113
+ yield* runTransaction("read Thread recovery deadlines", async () => {
1114
+ for (const threadId of new Set(current.map((row) => row.threadId))) {
1115
+ if (recovery.loaded.has(threadId)) continue;
1116
+ const encoded = await ctx.storage.get(recoveryFaultKey(threadId));
1117
+
1118
+ if (encoded !== undefined)
1119
+ recovery.faults.set(threadId, decodeRecoveryFault(threadId, encoded));
1120
+ recovery.loaded.add(threadId);
1121
+ }
1122
+ });
1123
+ const reports = recovery.reports;
1124
+ const recoveryFaults = recovery.faults;
1125
+
1126
+ const waiting = (row: SubmissionWorkItem) =>
1127
+ !recovery.pending.has(row.threadId) &&
1128
+ !recoveryFaults.has(row.threadId) &&
1129
+ stableExternalWait(row, reports);
1130
+
1131
+ const heads = new Map<ThreadId, SubmissionWorkItem>();
871
1132
 
872
1133
  for (const row of current) {
873
- // Parked uncertainty keeps its settlement obligation, but later input can run.
874
- // Accepted aborts and every other wait remain subject to the lane's FIFO barrier.
875
- if (row.state === "unknown" && stableExternalWait(row, reports)) continue;
1134
+ // Only recovered uncertainty may release later input; accepted aborts retain FIFO.
1135
+ if (row.state === "unknown" && waiting(row)) continue;
876
1136
  if (!heads.has(row.threadId)) heads.set(row.threadId, row);
877
1137
  }
1138
+ const stopping = new Set<ThreadId>();
1139
+
1140
+ for (const head of heads.values()) {
1141
+ if (
1142
+ head.state !== "ready" ||
1143
+ recovery.pending.has(head.threadId) ||
1144
+ recoveryFaults.has(head.threadId)
1145
+ )
1146
+ continue;
1147
+
1148
+ // An accepted abort is cleanup even when its input was never claimed. This
1149
+ // control-only read must not decode the execution journal or a recovery snapshot.
1150
+ const intent = yield* isolateRecovery(
1151
+ ledger.readAbortIntent(AbortIntentRequest.make({ submissionId: head.submissionId })),
1152
+ {
1153
+ timeout: recoveryConfig.recoveryTimeout,
1154
+ phase: () => "recovery",
1155
+ operation: "read abort intent",
1156
+ },
1157
+ );
1158
+
1159
+ if (Result.isSuccess(intent)) {
1160
+ if (intent.success !== undefined) stopping.add(head.threadId);
1161
+ } else {
1162
+ const faults = yield* recordRecoveryStatus(
1163
+ RecoverySweepResult.make({
1164
+ reports: [],
1165
+ blocked: [
1166
+ RecoveryBlocked.make({ threadId: head.threadId, failure: intent.failure }),
1167
+ ],
1168
+ }),
1169
+ );
1170
+
1171
+ for (const [threadId, fault] of faults) recoveryFaults.set(threadId, fault);
1172
+ recovery.recovered++;
1173
+ }
1174
+ }
878
1175
 
879
1176
  const eligible = [...heads.values()]
880
- .filter((head) => !stableExternalWait(head, reports))
1177
+ .filter(
1178
+ (head) =>
1179
+ !recovery.pending.has(head.threadId) &&
1180
+ !stopping.has(head.threadId) &&
1181
+ !recoveryFaults.has(head.threadId) &&
1182
+ !waiting(head) &&
1183
+ (head.state === "ready" || reports.has(head.submissionId)),
1184
+ )
881
1185
  .map((head) => head.threadId)
882
1186
  .sort();
883
1187
 
884
- const selectionTime = yield* Clock.currentTimeMillis;
885
-
886
1188
  yield* failpoint.hit("maintenance:select:before");
887
1189
 
888
1190
  const selection = yield* runTransaction("select maintenance lane", () =>
@@ -900,11 +1202,12 @@ export class ThreadMaintenance extends Context.Service<
900
1202
  ),
901
1203
  );
902
1204
 
903
- const next =
904
- runnable.find(
905
- (threadId) =>
906
- state.lastServedThreadId === undefined || threadId > state.lastServedThreadId,
907
- ) ?? runnable[0];
1205
+ const next = dispatch
1206
+ ? (runnable.find(
1207
+ (threadId) =>
1208
+ state.lastServedThreadId === undefined || threadId > state.lastServedThreadId,
1209
+ ) ?? runnable[0])
1210
+ : undefined;
908
1211
 
909
1212
  if (next !== undefined) {
910
1213
  await transaction.put(
@@ -915,22 +1218,54 @@ export class ThreadMaintenance extends Context.Service<
915
1218
  );
916
1219
  }
917
1220
 
918
- return { selected: next, retries };
1221
+ const backlog = recovery.started
1222
+ ? []
1223
+ : [...heads.keys()].filter((threadId) => {
1224
+ const fault = recoveryFaults.get(threadId);
1225
+
1226
+ return (
1227
+ !eligible.includes(threadId) &&
1228
+ (fault === undefined || fault.retryAt <= selectionTime)
1229
+ );
1230
+ });
1231
+
1232
+ // One finite wave, one Thread at a time, with the runtime's per-Thread deadline.
1233
+ // This cursor ensures an event deadline/eviction cannot always restart at the front.
1234
+ const after = backlog.filter(
1235
+ (threadId) =>
1236
+ state.lastRecoveredThreadId === undefined || threadId > state.lastRecoveredThreadId,
1237
+ );
1238
+
1239
+ const before = backlog.filter(
1240
+ (threadId) =>
1241
+ state.lastRecoveredThreadId !== undefined &&
1242
+ threadId <= state.lastRecoveredThreadId,
1243
+ );
1244
+
1245
+ return { selected: next, retries, backlog: [...after, ...before] };
919
1246
  }),
920
1247
  );
921
1248
 
922
1249
  yield* failpoint.hit("maintenance:select:after");
1250
+ if (!recovery.started) {
1251
+ recovery.started = true;
1252
+ recovery.needsCheckpoint = selection.backlog.length > 0;
1253
+ for (const threadId of selection.backlog) recovery.pending.add(threadId);
1254
+ yield* Deferred.succeed(recovery.queue, selection.backlog);
1255
+ }
923
1256
 
924
1257
  const selected =
925
1258
  selection.selected === undefined ? undefined : heads.get(selection.selected);
926
1259
 
1260
+ if (selected !== undefined) yield* recoverThread(selected.threadId, recovery);
1261
+
927
1262
  let retries = selection.retries;
928
1263
  let bindingFailure: DurableBindingFailure | undefined;
929
1264
 
930
1265
  // One runnable FIFO head per native opportunity. An absent agent waits for a deployment,
931
1266
  // including for children; other local lanes and host deliveries remain independently due.
932
1267
  const settlement =
933
- selected === undefined
1268
+ selected === undefined || recoveryFaults.has(selected.threadId)
934
1269
  ? Option.none()
935
1270
  : yield* runtime.processThreadHead(selected.threadId, { yieldAfter }).pipe(
936
1271
  Effect.catchTag("BindingUnavailable", (failure) => {
@@ -999,11 +1334,10 @@ export class ThreadMaintenance extends Context.Service<
999
1334
  const waitingHeads = new Map<ThreadId, boolean>();
1000
1335
 
1001
1336
  const autonomous = remaining.some((snapshot) => {
1002
- if (snapshot.state === "unknown" && stableExternalWait(snapshot, reports)) return false;
1337
+ if (snapshot.state === "unknown" && waiting(snapshot)) return false;
1003
1338
  const headWaiting = waitingHeads.get(snapshot.threadId);
1004
1339
 
1005
- if (headWaiting === undefined)
1006
- waitingHeads.set(snapshot.threadId, stableExternalWait(snapshot, reports));
1340
+ if (headWaiting === undefined) waitingHeads.set(snapshot.threadId, waiting(snapshot));
1007
1341
  // FIFO followers cannot execute through a stable external wait. Only plain queued
1008
1342
  // input is dormant here; admission repairs and accepted aborts still need a pass.
1009
1343
  if (
@@ -1013,26 +1347,28 @@ export class ThreadMaintenance extends Context.Service<
1013
1347
  )
1014
1348
  return false;
1015
1349
 
1016
- return !stableExternalWait(snapshot, reports);
1350
+ return !waiting(snapshot);
1017
1351
  });
1018
1352
 
1019
- const progressed =
1020
- Option.isSome(settlement) ||
1021
- recovered.some((report) => report.disposition === "repaired");
1353
+ const progressed = Option.isSome(settlement) || recovery.repaired;
1022
1354
 
1023
1355
  const now = yield* Clock.currentTimeMillis;
1024
1356
  const ordinaryDelay = autonomous ? yield* rearmDelay(progressed, started.stalls) : 0;
1025
1357
 
1026
- const nextEligible = eligible.map(
1027
- (threadId) =>
1028
- retries.find((retry) => retry.submissionId === heads.get(threadId)?.submissionId)
1029
- ?.notBefore ?? now,
1030
- );
1358
+ const nextEligible = [...recoveryFaults.values()]
1359
+ .map((fault) => fault.retryAt)
1360
+ .concat(
1361
+ eligible.map(
1362
+ (threadId) =>
1363
+ retries.find((retry) => retry.submissionId === heads.get(threadId)?.submissionId)
1364
+ ?.notBefore ?? now,
1365
+ ),
1366
+ );
1031
1367
 
1032
- const bindingDelay =
1368
+ const retryDelay =
1033
1369
  nextEligible.length === 0 ? 0 : Math.max(0, Math.min(...nextEligible) - now);
1034
1370
 
1035
- const delay = Math.max(ordinaryDelay, bindingDelay);
1371
+ const delay = Math.max(ordinaryDelay, retryDelay);
1036
1372
 
1037
1373
  // Checkpoint native progress without changing the physical alarm. Auxiliary
1038
1374
  // delivery remains live; later mutations still advance the shared generation.
@@ -1044,11 +1380,14 @@ export class ThreadMaintenance extends Context.Service<
1044
1380
  const { state } = await readMaintenanceState(transaction);
1045
1381
 
1046
1382
  const processed =
1047
- autonomous || started.activeAtStart > 0 || active > 0
1383
+ autonomous ||
1384
+ observation.activeAtStart > 0 ||
1385
+ started.activeAtStart > 0 ||
1386
+ active > 0
1048
1387
  ? state.processed
1049
- : state.processed > started.generation
1388
+ : state.processed > observation.generation
1050
1389
  ? state.processed
1051
- : started.generation;
1390
+ : observation.generation;
1052
1391
 
1053
1392
  const next = ThreadMaintenanceState.make({
1054
1393
  ...Struct.omit(state, ["retry"]),
@@ -1087,7 +1426,6 @@ export class ThreadMaintenance extends Context.Service<
1087
1426
 
1088
1427
  return {
1089
1428
  phase: "actionable",
1090
- recovered: recovered.length,
1091
1429
  settled: Option.isSome(settlement) ? 1 : 0,
1092
1430
  nonterminal: remaining.length,
1093
1431
  nextAttemptAt,
@@ -1106,7 +1444,9 @@ export class ThreadMaintenance extends Context.Service<
1106
1444
  Effect.catch(() => Effect.never),
1107
1445
  );
1108
1446
 
1109
- let started = yield* beginNative(observed);
1447
+ // A wake may observe temporary backoff while this event's recovery is still running.
1448
+ // Its completion must retain the original actionable observation for acknowledgement.
1449
+ const started = yield* beginNative(observed);
1110
1450
 
1111
1451
  // This scope owns auxiliary dispatch and listeners, independently of native progress.
1112
1452
  // Close it before final alarm rearming, including on failure or event interruption.
@@ -1116,11 +1456,127 @@ export class ThreadMaintenance extends Context.Service<
1116
1456
 
1117
1457
  const dispatchClosed = yield* Deferred.make<void>();
1118
1458
  const stopDispatch = Deferred.await(dispatchClosed);
1459
+ // Coalesce local scan hints while a lane is busy; one hint requests a current read.
1460
+ const checks = yield* PubSub.sliding<void>(1);
1461
+
1462
+ yield* Effect.addFinalizer(() => PubSub.shutdown(checks));
1463
+ let dispatchOpen = true;
1464
+ let activityChanged = yield* Deferred.make<void>();
1465
+ const signalActivity = Effect.suspend(() => Deferred.succeed(activityChanged, undefined));
1466
+
1467
+ const closeDispatch = Effect.sync(() => {
1468
+ dispatchOpen = false;
1469
+ }).pipe(Effect.andThen(Deferred.succeed(dispatchClosed, undefined)));
1470
+
1471
+ const makeActivity = Effect.fnUntraced(function* (allowance?: number) {
1472
+ const initialized = yield* Deferred.make<void>();
1473
+ let active = 0;
1474
+ let failed = false;
1475
+
1476
+ const ready = Deferred.succeed(initialized, undefined).pipe(
1477
+ Effect.andThen(signalActivity),
1478
+ Effect.asVoid,
1479
+ Effect.uninterruptible,
1480
+ );
1481
+
1482
+ const activity: ThreadMaintenanceActivity["Service"] = {
1483
+ ready,
1484
+ subscribeChanges: PubSub.subscribe(checks).pipe(Effect.map(PubSub.take)),
1485
+ run: (wave) =>
1486
+ Effect.acquireUseRelease(
1487
+ Effect.sync(() => {
1488
+ if (!dispatchOpen) return false;
1489
+ active++;
1490
+
1491
+ return true;
1492
+ }),
1493
+ (admitted) =>
1494
+ !admitted
1495
+ ? Effect.void
1496
+ : (allowance === undefined
1497
+ ? wave
1498
+ : wave.pipe(
1499
+ Effect.timeoutOrElse({
1500
+ duration: allowance,
1501
+ orElse: () =>
1502
+ DurableAlarmError.make({
1503
+ operation: "host dispatch allowance",
1504
+ message:
1505
+ "The admitted host wave exceeded its allowance; durable work remains pending",
1506
+ }),
1507
+ }),
1508
+ )
1509
+ ).pipe(
1510
+ Effect.onExit((exit) =>
1511
+ Effect.sync(() => {
1512
+ if (Exit.isFailure(exit)) failed = true;
1513
+ }),
1514
+ ),
1515
+ ),
1516
+ (admitted) =>
1517
+ Effect.sync(() => {
1518
+ if (admitted) active--;
1519
+ }).pipe(Effect.andThen(signalActivity)),
1520
+ ),
1521
+ };
1522
+
1523
+ return {
1524
+ activity,
1525
+ initialized: Deferred.await(initialized),
1526
+ quiet: () => active === 0 && Deferred.isDoneUnsafe(initialized),
1527
+ failed: () => failed,
1528
+ };
1529
+ });
1530
+
1531
+ const messageActivity = yield* makeActivity();
1532
+ const hostActivity = yield* makeActivity(host.dispatchTimeoutMillis);
1533
+
1534
+ const recovery: NativeRecovery = {
1535
+ queue: yield* Deferred.make<ReadonlyArray<ThreadId>>(),
1536
+ pending: new Set(),
1537
+ loaded: new Set(),
1538
+ reports: new Map(),
1539
+ faults: new Map(),
1540
+ started: false,
1541
+ needsCheckpoint: false,
1542
+ recovered: 0,
1543
+ repaired: false,
1544
+ };
1545
+
1546
+ const recoveryFiber = yield* Effect.forkIn(
1547
+ Effect.gen(function* () {
1548
+ for (const threadId of yield* Deferred.await(recovery.queue)) {
1549
+ yield* failpoint.hit("maintenance:select:before");
1550
+ yield* runTransaction("select old recovery lane", () =>
1551
+ ctx.storage.transaction(async (transaction) => {
1552
+ const { state } = await readMaintenanceState(transaction);
1553
+
1554
+ await transaction.put(
1555
+ MAINTENANCE_STATE_KEY,
1556
+ encodeMaintenanceState(
1557
+ ThreadMaintenanceState.make({ ...state, lastRecoveredThreadId: threadId }),
1558
+ ),
1559
+ );
1560
+ }),
1561
+ );
1562
+ yield* failpoint.hit("maintenance:select:after");
1563
+ yield* recoverThread(threadId, recovery);
1564
+ recovery.pending.delete(threadId);
1565
+ yield* wakes.notify(threadId);
1566
+ }
1567
+ }),
1568
+ auxiliaryScope,
1569
+ );
1119
1570
 
1120
1571
  // Fork setup too: an ordinary auxiliary setup failure is reported after native work,
1121
1572
  // rather than gating its opportunity. Event interruption still closes every fiber.
1122
1573
  const deliveryFiber = yield* Effect.forkIn(
1123
- Scope.provide(auxiliaryScope)(messages.drainUntil(stopDispatch, dispatchUntil)),
1574
+ Scope.provide(auxiliaryScope)(
1575
+ messages.drainUntil(stopDispatch, dispatchUntil).pipe(
1576
+ Effect.provideService(ThreadMaintenanceActivity, messageActivity.activity),
1577
+ Effect.onExit(() => messageActivity.activity.ready),
1578
+ ),
1579
+ ),
1124
1580
  auxiliaryScope,
1125
1581
  );
1126
1582
 
@@ -1137,8 +1593,29 @@ export class ThreadMaintenance extends Context.Service<
1137
1593
  }),
1138
1594
  ),
1139
1595
  );
1140
- yield* host.drainUntil(stopDispatch, dispatchUntil);
1141
- }),
1596
+
1597
+ // Initial setup must also be finite. Once it is accounted for, each actual
1598
+ // wave has its own unchanged allowance; passive listeners have no timer.
1599
+ const initialization = hostActivity.initialized.pipe(
1600
+ Effect.timeoutOrElse({
1601
+ duration: host.dispatchTimeoutMillis,
1602
+ orElse: () =>
1603
+ DurableAlarmError.make({
1604
+ operation: "host initialization",
1605
+ message:
1606
+ "The host did not account for initial maintenance before its allowance; durable work remains pending",
1607
+ }),
1608
+ }),
1609
+ Effect.andThen(Effect.never),
1610
+ );
1611
+
1612
+ yield* Effect.raceFirst(
1613
+ host
1614
+ .drainUntil(stopDispatch, dispatchUntil)
1615
+ .pipe(Effect.provideService(ThreadMaintenanceActivity, hostActivity.activity)),
1616
+ initialization,
1617
+ );
1618
+ }).pipe(Effect.onExit(() => hostActivity.activity.ready)),
1142
1619
  ),
1143
1620
  auxiliaryScope,
1144
1621
  );
@@ -1148,45 +1625,99 @@ export class ThreadMaintenance extends Context.Service<
1148
1625
  drainDue.pipe(
1149
1626
  Effect.provideService(ThreadProjectionMaintenance, projection),
1150
1627
  Effect.timeoutOption(config.projectionDispatchTimeoutMillis),
1628
+ Effect.onExit(() => signalActivity),
1151
1629
  ),
1152
1630
  auxiliaryScope,
1153
1631
  );
1154
1632
 
1155
- let result = yield* advance(started, yieldAfter, observed);
1633
+ let result = yield* advance(started, yieldAfter, observed, recovery);
1634
+
1635
+ // This event owns one finite old-recovery wave, including an empty caught-up wave.
1636
+ recovery.started = true;
1637
+ yield* Deferred.succeed(recovery.queue, []);
1156
1638
  let phase = result.phase;
1157
- let recovered = result.recovered;
1158
1639
  let settled = result.settled;
1159
1640
 
1160
1641
  observed.nativeOnly = false;
1161
1642
 
1162
- // Close admission of new delivery waves once, then keep advancing native
1163
- // work while the already-admitted waves finish. Neither lane restarts the
1164
- // other's work or receives a fresh event budget.
1165
- yield* Deferred.succeed(dispatchClosed, undefined);
1643
+ // Pump termination and quiescence are distinct: listeners await closure, while their
1644
+ // finite waves report activity. A held sibling never retires another lane's controls.
1645
+ const hostJoin = yield* Effect.forkIn(
1646
+ Effect.gen(function* () {
1647
+ yield* stopDispatch;
1166
1648
 
1167
- const remaining = Math.max(
1168
- 1,
1169
- DateTime.toEpochMillis(dispatchUntil) - (yield* Clock.currentTimeMillis),
1170
- );
1649
+ const remaining = Math.max(
1650
+ 1,
1651
+ DateTime.toEpochMillis(dispatchUntil) - (yield* Clock.currentTimeMillis),
1652
+ );
1171
1653
 
1172
- const hostJoin = yield* Effect.forkIn(
1173
- Fiber.join(hostFiber).pipe(
1174
- Effect.timeoutOption(Math.min(host.dispatchTimeoutMillis, remaining)),
1175
- Effect.tap((outcome) =>
1176
- Effect.annotateCurrentSpan({ "host.timedOut": Option.isNone(outcome) }),
1177
- ),
1178
- ),
1654
+ const outcome = yield* Fiber.join(hostFiber).pipe(
1655
+ Effect.timeoutOption(Math.min(host.dispatchTimeoutMillis, remaining)),
1656
+ );
1657
+
1658
+ yield* Effect.annotateCurrentSpan({ "host.timedOut": Option.isNone(outcome) });
1659
+ }),
1179
1660
  auxiliaryScope,
1180
1661
  );
1181
1662
 
1182
- const retired = yield* Effect.forkChild(Fiber.joinAll([deliveryFiber, hostJoin, backfill]));
1663
+ const retired = yield* Effect.forkChild(
1664
+ Fiber.joinAll([deliveryFiber, hostJoin, backfill, recoveryFiber]),
1665
+ );
1666
+
1667
+ while (true) {
1668
+ const recoveryFinished = recoveryFiber.pollUnsafe() !== undefined;
1669
+ const retiredExit = retired.pollUnsafe();
1670
+
1671
+ if (recoveryFinished) {
1672
+ if (recovery.needsCheckpoint) {
1673
+ yield* Fiber.join(recoveryFiber);
1674
+ // Fold the original recovery observation into its checkpoint before closing.
1675
+ result = yield* advance(started, yieldAfter, observed, recovery, settled === 0);
1676
+ if (result.phase === "actionable") phase = "actionable";
1677
+ settled += result.settled;
1678
+ recovery.needsCheckpoint = false;
1679
+ yield* PubSub.publish(checks, undefined);
1680
+ }
1681
+ }
1682
+ if (retiredExit !== undefined && Exit.isFailure(retiredExit)) break;
1183
1683
 
1184
- const auxiliaryPending = () =>
1185
- deliveryFiber.pollUnsafe() === undefined ||
1186
- hostFiber.pollUnsafe() === undefined ||
1187
- backfill.pollUnsafe() === undefined;
1684
+ if (
1685
+ recoveryFinished &&
1686
+ backfill.pollUnsafe() !== undefined &&
1687
+ messageActivity.quiet() &&
1688
+ hostActivity.quiet()
1689
+ ) {
1690
+ const now = yield* Clock.currentTimeMillis;
1691
+ const messageDeadline = yield* messages.pendingDeadline;
1692
+ const hostDeadline = yield* host.pendingDeadline;
1693
+
1694
+ const messageDue =
1695
+ deliveryFiber.pollUnsafe() === undefined &&
1696
+ !messageActivity.failed() &&
1697
+ Option.isSome(messageDeadline) &&
1698
+ messageDeadline.value <= now;
1699
+
1700
+ const hostDue =
1701
+ hostFiber.pollUnsafe() === undefined &&
1702
+ !hostActivity.failed() &&
1703
+ now + host.dispatchTimeoutMillis <= DateTime.toEpochMillis(dispatchUntil) &&
1704
+ Option.isSome(hostDeadline) &&
1705
+ hostDeadline.value <= now;
1706
+
1707
+ // No asynchronous operation separates the activity recheck from closure. A
1708
+ // racing registration either keeps this window open or cannot start its body.
1709
+ const closed = yield* Effect.sync(() => {
1710
+ if (messageDue || hostDue || !messageActivity.quiet() || !hostActivity.quiet())
1711
+ return false;
1712
+
1713
+ dispatchOpen = false;
1714
+
1715
+ return true;
1716
+ });
1188
1717
 
1189
- while (retired.pollUnsafe() === undefined && auxiliaryPending()) {
1718
+ if (closed) break;
1719
+ yield* PubSub.publish(checks, undefined);
1720
+ }
1190
1721
  const now = yield* Clock.currentTimeMillis;
1191
1722
  const until = DateTime.toEpochMillis(yieldAfter);
1192
1723
 
@@ -1198,24 +1729,50 @@ export class ThreadMaintenance extends Context.Service<
1198
1729
  until,
1199
1730
  );
1200
1731
 
1732
+ // A completion racing this iteration stays armed until the loop handles it.
1733
+ const recoveryDone = recoveryFinished ? Effect.never : Fiber.await(recoveryFiber);
1734
+ const retirementDone = retiredExit === undefined ? Fiber.await(retired) : Effect.never;
1735
+ const changed = activityChanged;
1736
+
1201
1737
  const ready = yield* Effect.raceFirst(
1202
- Effect.raceFirst(notified, Effect.sleep(Math.max(0, next - now))).pipe(Effect.as(true)),
1203
- Fiber.await(retired).pipe(Effect.as(false)),
1738
+ Effect.raceFirst(notified, Effect.sleep(Math.max(0, next - now))).pipe(
1739
+ Effect.as("native" as const),
1740
+ ),
1741
+ Effect.raceFirst(
1742
+ recoveryDone.pipe(Effect.as("recovery" as const)),
1743
+ Effect.raceFirst(
1744
+ retirementDone.pipe(Effect.as("retired" as const)),
1745
+ Deferred.await(changed).pipe(Effect.as("activity" as const)),
1746
+ ),
1747
+ ),
1204
1748
  );
1205
1749
 
1206
- if (!ready || retired.pollUnsafe() !== undefined || !auxiliaryPending()) break;
1750
+ if (ready === "recovery" || ready === "retired") continue;
1751
+ if (ready === "activity") activityChanged = yield* Deferred.make<void>();
1207
1752
  if ((yield* Clock.currentTimeMillis) >= until) break;
1753
+ if (ready === "native") yield* PubSub.publish(checks, undefined);
1208
1754
 
1209
- started = yield* beginNative(observed);
1210
- result = yield* advance(started, yieldAfter, observed);
1755
+ const awakened = yield* beginNative(observed);
1756
+
1757
+ result = yield* advance(awakened, yieldAfter, observed, recovery);
1211
1758
  if (result.phase === "actionable") phase = "actionable";
1212
- recovered += result.recovered;
1213
1759
  settled += result.settled;
1214
1760
  observed.nativeOnly = false;
1761
+ if (result.settled > 0) yield* PubSub.publish(checks, undefined);
1215
1762
  }
1763
+ // The original native yield deadline closes all new waves even if old recovery
1764
+ // is still pending. No recovery or delivery receives a renewed event budget.
1765
+ yield* closeDispatch;
1216
1766
  // Preserve driver-owned Claim deadlines and failures, then close every
1217
1767
  // listener before the one final alarm decision.
1218
1768
  yield* Fiber.join(retired);
1769
+ if (recovery.needsCheckpoint) {
1770
+ // The native yield deadline ended dispatch before old recovery finished. Fold its
1771
+ // control state into acknowledgement without starting an Attempt after retirement.
1772
+ result = yield* advance(started, yieldAfter, observed, recovery, false);
1773
+ if (result.phase === "actionable") phase = "actionable";
1774
+ settled += result.settled;
1775
+ }
1219
1776
  yield* Scope.close(auxiliaryScope, Exit.void);
1220
1777
  yield* failpoint.hit("maintenance:finish:before");
1221
1778
 
@@ -1256,7 +1813,7 @@ export class ThreadMaintenance extends Context.Service<
1256
1813
 
1257
1814
  const report = MaintenancePassReport.make({
1258
1815
  phase,
1259
- recovered,
1816
+ recovered: recovery.recovered,
1260
1817
  settled,
1261
1818
  nonterminal: result.nonterminal,
1262
1819
  alarm: disposition,
@@ -1309,6 +1866,7 @@ export class ThreadMaintenance extends Context.Service<
1309
1866
  }),
1310
1867
  ),
1311
1868
  ensureAlarm: mutations.withSnapshot(() => ensureAlarm()),
1869
+ recoveryStatus,
1312
1870
  withMutation: (body) =>
1313
1871
  mutations.withMutation(
1314
1872
  body.pipe(