@effect-agent/platform-cloudflare 0.1.0-beta.111 → 0.1.0-beta.113
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/Alarm.d.mts +65 -15
- package/dist/Alarm.mjs +326 -50
- package/dist/Alarm.mjs.map +1 -1
- package/dist/CloudflareThreadClient.d.mts +19 -12
- package/dist/CloudflareThreadClient.mjs +1 -1
- package/dist/{ThreadObject-BGla_Obk.mjs → ThreadObject-BPt5uByD.mjs} +18 -9
- package/dist/ThreadObject-BPt5uByD.mjs.map +1 -0
- package/dist/{ThreadObject-Cmn8pAQ3.d.mts → ThreadObject-DAuaRyGl.d.mts} +37 -30
- package/dist/ThreadObject.d.mts +1 -1
- package/dist/ThreadObject.mjs +1 -1
- package/dist/index.d.mts +1 -1
- package/dist/index.mjs +1 -1
- package/package.json +1 -1
- package/src/Alarm.ts +642 -84
- package/src/ThreadObject.ts +2 -2
- package/src/internal/message-delivery.ts +33 -23
- package/dist/ThreadObject-BGla_Obk.mjs.map +0 -1
package/src/Alarm.ts
CHANGED
|
@@ -9,8 +9,10 @@ import {
|
|
|
9
9
|
Fiber,
|
|
10
10
|
Layer,
|
|
11
11
|
Option,
|
|
12
|
+
PubSub,
|
|
12
13
|
Random,
|
|
13
14
|
Ref,
|
|
15
|
+
Result,
|
|
14
16
|
Schema,
|
|
15
17
|
Scope,
|
|
16
18
|
Semaphore,
|
|
@@ -20,11 +22,25 @@ import {
|
|
|
20
22
|
import { type DurableBindingFailure } from "effect-agent/agent-registration";
|
|
21
23
|
import {
|
|
22
24
|
DurableAgentRuntime,
|
|
25
|
+
DurableRuntimeConfig,
|
|
26
|
+
isolateRecovery,
|
|
27
|
+
RecoveryBlocked,
|
|
28
|
+
RecoveryFailure,
|
|
23
29
|
type DurableWorkerFailure,
|
|
24
30
|
type RecoveryReport,
|
|
31
|
+
RecoverySweepResult,
|
|
25
32
|
} from "effect-agent/durable-agent-runtime";
|
|
26
33
|
import { ThreadId, SubmissionId } from "effect-agent/identifiers";
|
|
27
|
-
import {
|
|
34
|
+
import {
|
|
35
|
+
OperationAuthorizationRequest,
|
|
36
|
+
OperationAuthorizer,
|
|
37
|
+
type OperationDenied,
|
|
38
|
+
} from "effect-agent/operation-authorizer";
|
|
39
|
+
import {
|
|
40
|
+
AbortIntentRequest,
|
|
41
|
+
SubmissionLedger,
|
|
42
|
+
type SubmissionWorkItem,
|
|
43
|
+
} from "effect-agent/submission-ledger";
|
|
28
44
|
import {
|
|
29
45
|
ThreadProjectionMaintenance,
|
|
30
46
|
drainDue,
|
|
@@ -187,7 +203,7 @@ export class MaintenancePassReport extends Schema.Class<MaintenancePassReport>(
|
|
|
187
203
|
)({
|
|
188
204
|
/** `caught-up` ran no runtime work (publication may be pending); `actionable` ran recovery. */
|
|
189
205
|
phase: Schema.Literals(["caught-up", "actionable"]),
|
|
190
|
-
/** Recovery decisions
|
|
206
|
+
/** Recovery decisions and Thread faults observed during this event. */
|
|
191
207
|
recovered: Schema.Int.check(Schema.isGreaterThanOrEqualTo(0)),
|
|
192
208
|
/** Head Attempts settled during the event. Joined input may settle with each head. */
|
|
193
209
|
settled: Schema.Int.check(Schema.isGreaterThanOrEqualTo(0)),
|
|
@@ -211,6 +227,8 @@ export type ThreadMaintenanceFailpointLocation =
|
|
|
211
227
|
| "maintenance:select:after"
|
|
212
228
|
| "maintenance:binding-retry:before"
|
|
213
229
|
| "maintenance:binding-retry:after"
|
|
230
|
+
| "maintenance:recovery-status:before"
|
|
231
|
+
| "maintenance:recovery-status:after"
|
|
214
232
|
| "maintenance:retry:before"
|
|
215
233
|
| "maintenance:retry:after"
|
|
216
234
|
| "maintenance:checkpoint:before"
|
|
@@ -268,6 +286,65 @@ export class ThreadPublication extends Context.Service<
|
|
|
268
286
|
});
|
|
269
287
|
}
|
|
270
288
|
|
|
289
|
+
/** Event-local accounting shared by the existing maintenance pumps and native scheduler. */
|
|
290
|
+
export class ThreadMaintenanceActivity extends Context.Service<
|
|
291
|
+
ThreadMaintenanceActivity,
|
|
292
|
+
{
|
|
293
|
+
/**
|
|
294
|
+
* Account for one finite selection/dispatch, through joined completion or interruption.
|
|
295
|
+
* Register before reading/claiming work. After dispatch closes the body is not started.
|
|
296
|
+
* Host waves retain their declared allowance from this call; message Claims keep the driver's
|
|
297
|
+
* own deadline. Waiting for notifications belongs outside this bracket.
|
|
298
|
+
*/
|
|
299
|
+
readonly run: <R>(
|
|
300
|
+
wave: Effect.Effect<void, DurableAlarmError, R>,
|
|
301
|
+
) => Effect.Effect<void, DurableAlarmError, R>;
|
|
302
|
+
/** Acknowledge all initial subscriptions/scans. Composite hosts use `all` below. */
|
|
303
|
+
readonly ready: Effect.Effect<void>;
|
|
304
|
+
/**
|
|
305
|
+
* Acquire an independent scoped subscription before the initial scan and `ready`.
|
|
306
|
+
* The returned wait observes coalesced native scan hints, including those published before
|
|
307
|
+
* its first execution. Each hint requests a current local due-work check.
|
|
308
|
+
*/
|
|
309
|
+
readonly subscribeChanges: Effect.Effect<Effect.Effect<void>, never, Scope.Scope>;
|
|
310
|
+
}
|
|
311
|
+
>()("@effect-agent/platform-cloudflare/ThreadMaintenanceActivity") {
|
|
312
|
+
/** Join child pumps; acknowledge the parent only after every child is ready or has exited. */
|
|
313
|
+
static readonly all = Effect.fnUntraced(function* <A, E, R>(
|
|
314
|
+
lanes: ReadonlyArray<Effect.Effect<A, E, R>>,
|
|
315
|
+
): Effect.fn.Return<ReadonlyArray<A>, E, R | ThreadMaintenanceActivity> {
|
|
316
|
+
const activity = yield* ThreadMaintenanceActivity;
|
|
317
|
+
let remaining = lanes.length;
|
|
318
|
+
|
|
319
|
+
if (remaining === 0) return yield* activity.ready.pipe(Effect.as([]));
|
|
320
|
+
|
|
321
|
+
return yield* Effect.forEach(
|
|
322
|
+
lanes,
|
|
323
|
+
(lane) => {
|
|
324
|
+
let ready = false;
|
|
325
|
+
|
|
326
|
+
const child: ThreadMaintenanceActivity["Service"] = {
|
|
327
|
+
...activity,
|
|
328
|
+
ready: Effect.suspend(() => {
|
|
329
|
+
if (ready) return Effect.void;
|
|
330
|
+
ready = true;
|
|
331
|
+
|
|
332
|
+
return --remaining === 0 ? activity.ready : Effect.void;
|
|
333
|
+
}).pipe(Effect.uninterruptible),
|
|
334
|
+
};
|
|
335
|
+
|
|
336
|
+
// A failed initial setup is accounted for without swallowing its Cause. Callers
|
|
337
|
+
// may collect Exits when sibling pumps must keep their independent opportunities.
|
|
338
|
+
return lane.pipe(
|
|
339
|
+
Effect.provideService(ThreadMaintenanceActivity, child),
|
|
340
|
+
Effect.onExit(() => child.ready),
|
|
341
|
+
);
|
|
342
|
+
},
|
|
343
|
+
{ concurrency: "unbounded" },
|
|
344
|
+
);
|
|
345
|
+
});
|
|
346
|
+
}
|
|
347
|
+
|
|
271
348
|
/**
|
|
272
349
|
* Host-assembled native message recovery. The driver bounds each actual Claim and persists its
|
|
273
350
|
* timeout/retry before this pump returns. Do not add a second timer starting at batch selection:
|
|
@@ -277,7 +354,7 @@ export const ThreadMessageDelivery = Context.Reference<{
|
|
|
277
354
|
readonly drainUntil: (
|
|
278
355
|
dispatchClosed: Effect.Effect<void>,
|
|
279
356
|
dispatchUntil: DateTime.Utc,
|
|
280
|
-
) => Effect.Effect<void, DurableAlarmError, Scope.Scope>;
|
|
357
|
+
) => Effect.Effect<void, DurableAlarmError, Scope.Scope | ThreadMaintenanceActivity>;
|
|
281
358
|
readonly pendingDeadline: Effect.Effect<Option.Option<number>, DurableAlarmError>;
|
|
282
359
|
}>("@effect-agent/platform-cloudflare/ThreadMessageDelivery", {
|
|
283
360
|
defaultValue: () => ({
|
|
@@ -288,14 +365,16 @@ export const ThreadMessageDelivery = Context.Reference<{
|
|
|
288
365
|
|
|
289
366
|
/**
|
|
290
367
|
* Application obligations sharing this Object's alarm. Admit one initial external wave even on
|
|
291
|
-
* a caught-up pass, then respond to wakes
|
|
292
|
-
*
|
|
293
|
-
*
|
|
368
|
+
* a caught-up pass, then respond to wakes and ThreadMaintenanceActivity.subscribeChanges until
|
|
369
|
+
* dispatchClosed. Yield the event's ThreadMaintenanceActivity to bracket each finite lane with
|
|
370
|
+
* run and acknowledge initial setup with ready.
|
|
371
|
+
* Passive subscriptions do not count as active work. Closure stops both new waves and native
|
|
372
|
+
* Attempts; local admission/control subscriptions belong to the event Scope until teardown.
|
|
294
373
|
* No deadline sleeps or automatic retry loops. Return after already-admitted waves finish.
|
|
295
374
|
*
|
|
296
375
|
* Declare a finite whole-wave allowance (1..300000ms): maximum for parallel lanes, sum for
|
|
297
376
|
* sequential operations. Admit a wave only if its allowance fits before dispatchUntil. Later
|
|
298
|
-
* arrivals cannot renew
|
|
377
|
+
* arrivals cannot renew an admitted wave's allowance. Maintenance bounds each registered wave,
|
|
299
378
|
* interrupts and joins event Scope, then reads local deadlines under the mutation gate.
|
|
300
379
|
*
|
|
301
380
|
* Setup and pendingDeadline are bounded local operations. Persist claims/envelopes before
|
|
@@ -309,7 +388,7 @@ export const ThreadHostMaintenance = Context.Reference<{
|
|
|
309
388
|
readonly drainUntil: (
|
|
310
389
|
dispatchClosed: Effect.Effect<void>,
|
|
311
390
|
dispatchUntil: DateTime.Utc,
|
|
312
|
-
) => Effect.Effect<void, DurableAlarmError, Scope.Scope>;
|
|
391
|
+
) => Effect.Effect<void, DurableAlarmError, Scope.Scope | ThreadMaintenanceActivity>;
|
|
313
392
|
readonly pendingDeadline: Effect.Effect<Option.Option<number>, DurableAlarmError>;
|
|
314
393
|
}>("@effect-agent/platform-cloudflare/ThreadHostMaintenance", {
|
|
315
394
|
defaultValue: () => ({
|
|
@@ -354,14 +433,64 @@ class BindingRetry extends Schema.Class<BindingRetry>("BindingRetry")({
|
|
|
354
433
|
reportedAt: Schema.Finite,
|
|
355
434
|
}) {}
|
|
356
435
|
|
|
436
|
+
/**
|
|
437
|
+
* A durable observation of blocked recovery, independent of the execution journal. This is
|
|
438
|
+
* neither a Settlement nor proof that external effects did not happen. A successful recovery
|
|
439
|
+
* sweep clears it; repair must preserve canonical history and the original admission identity.
|
|
440
|
+
*/
|
|
441
|
+
export class ThreadRecoveryFault extends Schema.Class<ThreadRecoveryFault>(
|
|
442
|
+
"@effect-agent/platform-cloudflare/ThreadRecoveryFault",
|
|
443
|
+
)({
|
|
444
|
+
schemaVersion: Schema.Literal(1),
|
|
445
|
+
threadId: ThreadId,
|
|
446
|
+
firstFailedAt: Schema.Finite,
|
|
447
|
+
lastFailedAt: Schema.Finite,
|
|
448
|
+
/** Earliest automatic recovery retry; new admissions do not erase this deadline. */
|
|
449
|
+
retryAt: Schema.Finite,
|
|
450
|
+
/** Saturates at 2^31 - 1; one observation per Thread per recovery sweep. */
|
|
451
|
+
attempts: Schema.Int.check(Schema.isGreaterThan(0), Schema.isLessThanOrEqualTo(2_147_483_647)),
|
|
452
|
+
failure: RecoveryFailure,
|
|
453
|
+
}) {}
|
|
454
|
+
|
|
455
|
+
const recoveryFaultKey = (threadId: ThreadId) =>
|
|
456
|
+
`effect-agent:thread-recovery-fault:v1:${threadId}`;
|
|
457
|
+
|
|
458
|
+
const decodeRecoveryFaultValue = Schema.decodeUnknownSync(ThreadRecoveryFault);
|
|
459
|
+
|
|
460
|
+
const decodeRecoveryFault = (threadId: ThreadId, encoded: unknown) => {
|
|
461
|
+
const fault = decodeRecoveryFaultValue(encoded);
|
|
462
|
+
|
|
463
|
+
if (fault.threadId !== threadId) throw new Error("Recovery status does not match its Thread key");
|
|
464
|
+
|
|
465
|
+
return fault;
|
|
466
|
+
};
|
|
467
|
+
|
|
468
|
+
const encodeRecoveryFault = Schema.encodeSync(ThreadRecoveryFault);
|
|
469
|
+
|
|
357
470
|
interface NativePassResult {
|
|
358
471
|
readonly phase: "caught-up" | "actionable";
|
|
359
|
-
readonly recovered: number;
|
|
360
472
|
readonly settled: number;
|
|
361
473
|
readonly nonterminal: number;
|
|
362
474
|
readonly nextAttemptAt: number | undefined;
|
|
363
475
|
}
|
|
364
476
|
|
|
477
|
+
/** Event-local observations only; durable ingress keeps racing mutations dirty. */
|
|
478
|
+
interface NativeRecovery {
|
|
479
|
+
readonly queue: Deferred.Deferred<ReadonlyArray<ThreadId>>;
|
|
480
|
+
readonly pending: Set<ThreadId>;
|
|
481
|
+
readonly loaded: Set<ThreadId>;
|
|
482
|
+
readonly reports: Map<SubmissionId, RecoveryReport>;
|
|
483
|
+
readonly faults: Map<ThreadId, ThreadRecoveryFault>;
|
|
484
|
+
observation?: {
|
|
485
|
+
readonly generation: bigint;
|
|
486
|
+
readonly activeAtStart: number;
|
|
487
|
+
};
|
|
488
|
+
started: boolean;
|
|
489
|
+
needsCheckpoint: boolean;
|
|
490
|
+
recovered: number;
|
|
491
|
+
repaired: boolean;
|
|
492
|
+
}
|
|
493
|
+
|
|
365
494
|
interface MaintenanceObservation {
|
|
366
495
|
generation?: bigint;
|
|
367
496
|
nativeOnly: boolean;
|
|
@@ -384,6 +513,8 @@ class ThreadMaintenanceState extends Schema.Class<ThreadMaintenanceState>(
|
|
|
384
513
|
nonterminal: Schema.Int.check(Schema.isGreaterThanOrEqualTo(0)),
|
|
385
514
|
/** One physical-owner cursor; old single-lane records need no conversion. */
|
|
386
515
|
lastServedThreadId: Schema.optionalKey(ThreadId),
|
|
516
|
+
/** Rotate old recovery independently of dispatch, including after eviction or timeout. */
|
|
517
|
+
lastRecoveredThreadId: Schema.optionalKey(ThreadId),
|
|
387
518
|
bindingRetries: Schema.optionalKey(Schema.Array(BindingRetry)),
|
|
388
519
|
/** Absent on older records. A newer mutation makes this retry obsolete. */
|
|
389
520
|
retry: Schema.optionalKey(MaintenanceRetry),
|
|
@@ -425,7 +556,7 @@ const ensureTransactionAlarmBy = async (
|
|
|
425
556
|
};
|
|
426
557
|
|
|
427
558
|
const stableExternalWait = (
|
|
428
|
-
snapshot:
|
|
559
|
+
snapshot: SubmissionWorkItem,
|
|
429
560
|
reports: ReadonlyMap<string, RecoveryReport>,
|
|
430
561
|
): boolean => {
|
|
431
562
|
const report = reports.get(snapshot.submissionId);
|
|
@@ -553,10 +684,10 @@ export type MaintenancePassFailure =
|
|
|
553
684
|
* recovery, ledger scans or canonical-history reads.
|
|
554
685
|
* 2. Reconcile before each head Attempt, then checkpoint only the observed generation. A racing
|
|
555
686
|
* producer keeps its newer generation dirty. Native retries retain their durable backoff.
|
|
556
|
-
* 3.
|
|
557
|
-
*
|
|
558
|
-
*
|
|
559
|
-
* 4. Native message delivery retains its driver-owned Claim deadline. Host/backfill
|
|
687
|
+
* 3. Keep native and delivery admission open together while finite waves remain active, so
|
|
688
|
+
* fresh replies and abort controls can progress during unrelated cleanup. Close atomically
|
|
689
|
+
* at quiescence, or at the original ten-minute yield deadline, before retiring listeners.
|
|
690
|
+
* 4. Native message delivery retains its driver-owned Claim deadline. Host/backfill waves are
|
|
560
691
|
* bounded independently; incoming native work never restarts or cancels their attempts.
|
|
561
692
|
* Auxiliary failures are reported after the current native opportunity.
|
|
562
693
|
* 5. Close every event resource before the final gated deadline snapshot and alarm decision.
|
|
@@ -572,6 +703,15 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
572
703
|
* generation has an alarm. It never scans the ledger or canonical history.
|
|
573
704
|
*/
|
|
574
705
|
readonly ensureAlarm: Effect.Effect<void, MaintenancePassFailure>;
|
|
706
|
+
/**
|
|
707
|
+
* Authorize `explain`, then read one bounded local record without reading execution history.
|
|
708
|
+
* None means no recorded fault, not proof of health or settlement. The host authenticates
|
|
709
|
+
* callers and verifies local Thread membership before exposing this service across RPC.
|
|
710
|
+
* Provide OperationAuthorizer when constructing this Layer, as for DurableAgentRuntime.
|
|
711
|
+
*/
|
|
712
|
+
readonly recoveryStatus: (
|
|
713
|
+
threadId: ThreadId,
|
|
714
|
+
) => Effect.Effect<Option.Option<ThreadRecoveryFault>, DurableAlarmError | OperationDenied>;
|
|
575
715
|
/**
|
|
576
716
|
* Serialize the pre-arm boundary with pass acknowledgement, advance the durable dirty
|
|
577
717
|
* generation and arm the alarm in one transaction BEFORE running the caller's mutation.
|
|
@@ -589,6 +729,7 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
589
729
|
| ThreadPublication
|
|
590
730
|
| ThreadProjectionMaintenance
|
|
591
731
|
| DurableAgentRuntime
|
|
732
|
+
| DurableRuntimeConfig
|
|
592
733
|
| SubmissionLedger
|
|
593
734
|
| WakeScheduler
|
|
594
735
|
| DurableAlarmService
|
|
@@ -599,6 +740,7 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
599
740
|
> = Layer.effect(ThreadMaintenance)(
|
|
600
741
|
Effect.gen(function* () {
|
|
601
742
|
const runtime = yield* DurableAgentRuntime;
|
|
743
|
+
const recoveryConfig = yield* DurableRuntimeConfig;
|
|
602
744
|
const ledger = yield* SubmissionLedger;
|
|
603
745
|
const wakes = yield* WakeScheduler;
|
|
604
746
|
const alarm = yield* DurableAlarmService;
|
|
@@ -611,6 +753,7 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
611
753
|
const projection = yield* ThreadProjectionMaintenance;
|
|
612
754
|
const messages = yield* ThreadMessageDelivery;
|
|
613
755
|
const host = yield* ThreadHostMaintenance;
|
|
756
|
+
const authorizer = yield* OperationAuthorizer;
|
|
614
757
|
|
|
615
758
|
// A broken disposable index still needs a retry alarm and must not prevent startup.
|
|
616
759
|
const projectionDeadline = projection.pendingDeadline.pipe(
|
|
@@ -635,6 +778,97 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
635
778
|
|
|
636
779
|
const runTransaction = yield* makeStorageOperation;
|
|
637
780
|
|
|
781
|
+
const recoveryStatus = Effect.fn("ThreadMaintenance.recoveryStatus")(function* (
|
|
782
|
+
threadId: ThreadId,
|
|
783
|
+
) {
|
|
784
|
+
yield* authorizer.authorize(
|
|
785
|
+
OperationAuthorizationRequest.make({ operation: "explain", threadId }),
|
|
786
|
+
);
|
|
787
|
+
|
|
788
|
+
return yield* runTransaction("read Thread recovery status", async () => {
|
|
789
|
+
const encoded = await ctx.storage.get(recoveryFaultKey(threadId));
|
|
790
|
+
|
|
791
|
+
return encoded === undefined
|
|
792
|
+
? Option.none()
|
|
793
|
+
: Option.some(decodeRecoveryFault(threadId, encoded));
|
|
794
|
+
});
|
|
795
|
+
});
|
|
796
|
+
|
|
797
|
+
const recordRecoveryStatus = Effect.fn("ThreadMaintenance.recordRecoveryStatus")(function* (
|
|
798
|
+
result: RecoverySweepResult,
|
|
799
|
+
) {
|
|
800
|
+
const threads = new Map<ThreadId, RecoveryFailure | undefined>();
|
|
801
|
+
|
|
802
|
+
for (const report of result.reports) threads.set(report.threadId, undefined);
|
|
803
|
+
for (const blocked of result.blocked) threads.set(blocked.threadId, blocked.failure);
|
|
804
|
+
if (threads.size === 0) return new Map<ThreadId, ThreadRecoveryFault>();
|
|
805
|
+
const now = yield* Clock.currentTimeMillis;
|
|
806
|
+
|
|
807
|
+
yield* failpoint.hit("maintenance:recovery-status:before");
|
|
808
|
+
|
|
809
|
+
const retained = yield* runTransaction("record Thread recovery status", () =>
|
|
810
|
+
ctx.storage.transaction(async (transaction) => {
|
|
811
|
+
const newlyBlocked: Array<ThreadRecoveryFault> = [];
|
|
812
|
+
const faults = new Map<ThreadId, ThreadRecoveryFault>();
|
|
813
|
+
|
|
814
|
+
for (const [threadId, failure] of threads) {
|
|
815
|
+
const key = recoveryFaultKey(threadId);
|
|
816
|
+
const encoded = await transaction.get(key);
|
|
817
|
+
|
|
818
|
+
const previous =
|
|
819
|
+
encoded === undefined ? undefined : decodeRecoveryFault(threadId, encoded);
|
|
820
|
+
|
|
821
|
+
if (failure === undefined) {
|
|
822
|
+
if (previous !== undefined) await transaction.delete(key);
|
|
823
|
+
continue;
|
|
824
|
+
}
|
|
825
|
+
|
|
826
|
+
const fault = ThreadRecoveryFault.make({
|
|
827
|
+
schemaVersion: 1,
|
|
828
|
+
threadId,
|
|
829
|
+
firstFailedAt: previous?.firstFailedAt ?? now,
|
|
830
|
+
lastFailedAt: now,
|
|
831
|
+
attempts: Math.min(2_147_483_647, (previous?.attempts ?? 0) + 1),
|
|
832
|
+
retryAt: now + Math.min(60_000, 5_000 * 2 ** Math.min(30, previous?.attempts ?? 0)),
|
|
833
|
+
failure,
|
|
834
|
+
});
|
|
835
|
+
|
|
836
|
+
await transaction.put(key, encodeRecoveryFault(fault));
|
|
837
|
+
faults.set(threadId, fault);
|
|
838
|
+
if (previous === undefined) newlyBlocked.push(fault);
|
|
839
|
+
}
|
|
840
|
+
|
|
841
|
+
return { newlyBlocked, faults };
|
|
842
|
+
}),
|
|
843
|
+
);
|
|
844
|
+
|
|
845
|
+
yield* failpoint.hit("maintenance:recovery-status:after");
|
|
846
|
+
for (const fault of retained.newlyBlocked)
|
|
847
|
+
yield* Effect.logError(
|
|
848
|
+
"Native Thread recovery blocked; accepted work remains pending",
|
|
849
|
+
fault.failure.reason === "defect"
|
|
850
|
+
? Cause.die(fault.failure)
|
|
851
|
+
: Cause.fail(fault.failure),
|
|
852
|
+
).pipe(Effect.annotateLogs({ threadId: fault.threadId }));
|
|
853
|
+
|
|
854
|
+
return retained.faults;
|
|
855
|
+
});
|
|
856
|
+
|
|
857
|
+
const recoverThread = Effect.fn("ThreadMaintenance.recoverThread")(function* (
|
|
858
|
+
threadId: ThreadId,
|
|
859
|
+
recovery: NativeRecovery,
|
|
860
|
+
) {
|
|
861
|
+
const result = yield* runtime.runRecovery({ threadId });
|
|
862
|
+
// Visibility is committed before a claim or any fallible auxiliary join.
|
|
863
|
+
const faults = yield* recordRecoveryStatus(result);
|
|
864
|
+
|
|
865
|
+
for (const report of result.reports) recovery.reports.set(report.submissionId, report);
|
|
866
|
+
recovery.faults.delete(threadId);
|
|
867
|
+
for (const [id, fault] of faults) recovery.faults.set(id, fault);
|
|
868
|
+
recovery.recovered += result.reports.length + result.blocked.length;
|
|
869
|
+
recovery.repaired ||= result.reports.some((report) => report.disposition === "repaired");
|
|
870
|
+
});
|
|
871
|
+
|
|
638
872
|
const ensureAlarm = Effect.fn("ThreadMaintenance.ensureAlarm")(function* () {
|
|
639
873
|
yield* failpoint.hit("maintenance:ensure:before");
|
|
640
874
|
const now = yield* Clock.currentTimeMillis;
|
|
@@ -821,6 +1055,8 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
821
1055
|
started: Effect.Success<ReturnType<typeof beginNative>>,
|
|
822
1056
|
yieldAfter: DateTime.Utc,
|
|
823
1057
|
observed: MaintenanceObservation,
|
|
1058
|
+
recovery: NativeRecovery,
|
|
1059
|
+
dispatch = true,
|
|
824
1060
|
): Effect.fn.Return<NativePassResult, MaintenancePassFailure> {
|
|
825
1061
|
const deadline = yield* publication.pendingDeadline;
|
|
826
1062
|
|
|
@@ -856,33 +1092,99 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
856
1092
|
|
|
857
1093
|
return {
|
|
858
1094
|
phase: "caught-up",
|
|
859
|
-
recovered: 0,
|
|
860
1095
|
settled: 0,
|
|
861
1096
|
nonterminal: started.nonterminal,
|
|
862
1097
|
nextAttemptAt,
|
|
863
1098
|
};
|
|
864
1099
|
}
|
|
865
|
-
//
|
|
1100
|
+
// Select from control state before reading execution history. A recovering or faulted
|
|
1101
|
+
// Thread cannot enter dispatch; old cleanup has its own scoped opportunity below.
|
|
866
1102
|
observed.nativeOnly = true;
|
|
867
|
-
|
|
868
|
-
|
|
1103
|
+
|
|
1104
|
+
// Every checkpoint keeps the producer overlap that belongs to this recovery wave.
|
|
1105
|
+
const observation = (recovery.observation ??= {
|
|
1106
|
+
generation: started.generation,
|
|
1107
|
+
activeAtStart: started.activeAtStart,
|
|
1108
|
+
});
|
|
1109
|
+
|
|
869
1110
|
const current = yield* Stream.runCollect(ledger.scanNonterminal);
|
|
870
|
-
const
|
|
1111
|
+
const selectionTime = yield* Clock.currentTimeMillis;
|
|
1112
|
+
|
|
1113
|
+
yield* runTransaction("read Thread recovery deadlines", async () => {
|
|
1114
|
+
for (const threadId of new Set(current.map((row) => row.threadId))) {
|
|
1115
|
+
if (recovery.loaded.has(threadId)) continue;
|
|
1116
|
+
const encoded = await ctx.storage.get(recoveryFaultKey(threadId));
|
|
1117
|
+
|
|
1118
|
+
if (encoded !== undefined)
|
|
1119
|
+
recovery.faults.set(threadId, decodeRecoveryFault(threadId, encoded));
|
|
1120
|
+
recovery.loaded.add(threadId);
|
|
1121
|
+
}
|
|
1122
|
+
});
|
|
1123
|
+
const reports = recovery.reports;
|
|
1124
|
+
const recoveryFaults = recovery.faults;
|
|
1125
|
+
|
|
1126
|
+
const waiting = (row: SubmissionWorkItem) =>
|
|
1127
|
+
!recovery.pending.has(row.threadId) &&
|
|
1128
|
+
!recoveryFaults.has(row.threadId) &&
|
|
1129
|
+
stableExternalWait(row, reports);
|
|
1130
|
+
|
|
1131
|
+
const heads = new Map<ThreadId, SubmissionWorkItem>();
|
|
871
1132
|
|
|
872
1133
|
for (const row of current) {
|
|
873
|
-
//
|
|
874
|
-
|
|
875
|
-
if (row.state === "unknown" && stableExternalWait(row, reports)) continue;
|
|
1134
|
+
// Only recovered uncertainty may release later input; accepted aborts retain FIFO.
|
|
1135
|
+
if (row.state === "unknown" && waiting(row)) continue;
|
|
876
1136
|
if (!heads.has(row.threadId)) heads.set(row.threadId, row);
|
|
877
1137
|
}
|
|
1138
|
+
const stopping = new Set<ThreadId>();
|
|
1139
|
+
|
|
1140
|
+
for (const head of heads.values()) {
|
|
1141
|
+
if (
|
|
1142
|
+
head.state !== "ready" ||
|
|
1143
|
+
recovery.pending.has(head.threadId) ||
|
|
1144
|
+
recoveryFaults.has(head.threadId)
|
|
1145
|
+
)
|
|
1146
|
+
continue;
|
|
1147
|
+
|
|
1148
|
+
// An accepted abort is cleanup even when its input was never claimed. This
|
|
1149
|
+
// control-only read must not decode the execution journal or a recovery snapshot.
|
|
1150
|
+
const intent = yield* isolateRecovery(
|
|
1151
|
+
ledger.readAbortIntent(AbortIntentRequest.make({ submissionId: head.submissionId })),
|
|
1152
|
+
{
|
|
1153
|
+
timeout: recoveryConfig.recoveryTimeout,
|
|
1154
|
+
phase: () => "recovery",
|
|
1155
|
+
operation: "read abort intent",
|
|
1156
|
+
},
|
|
1157
|
+
);
|
|
1158
|
+
|
|
1159
|
+
if (Result.isSuccess(intent)) {
|
|
1160
|
+
if (intent.success !== undefined) stopping.add(head.threadId);
|
|
1161
|
+
} else {
|
|
1162
|
+
const faults = yield* recordRecoveryStatus(
|
|
1163
|
+
RecoverySweepResult.make({
|
|
1164
|
+
reports: [],
|
|
1165
|
+
blocked: [
|
|
1166
|
+
RecoveryBlocked.make({ threadId: head.threadId, failure: intent.failure }),
|
|
1167
|
+
],
|
|
1168
|
+
}),
|
|
1169
|
+
);
|
|
1170
|
+
|
|
1171
|
+
for (const [threadId, fault] of faults) recoveryFaults.set(threadId, fault);
|
|
1172
|
+
recovery.recovered++;
|
|
1173
|
+
}
|
|
1174
|
+
}
|
|
878
1175
|
|
|
879
1176
|
const eligible = [...heads.values()]
|
|
880
|
-
.filter(
|
|
1177
|
+
.filter(
|
|
1178
|
+
(head) =>
|
|
1179
|
+
!recovery.pending.has(head.threadId) &&
|
|
1180
|
+
!stopping.has(head.threadId) &&
|
|
1181
|
+
!recoveryFaults.has(head.threadId) &&
|
|
1182
|
+
!waiting(head) &&
|
|
1183
|
+
(head.state === "ready" || reports.has(head.submissionId)),
|
|
1184
|
+
)
|
|
881
1185
|
.map((head) => head.threadId)
|
|
882
1186
|
.sort();
|
|
883
1187
|
|
|
884
|
-
const selectionTime = yield* Clock.currentTimeMillis;
|
|
885
|
-
|
|
886
1188
|
yield* failpoint.hit("maintenance:select:before");
|
|
887
1189
|
|
|
888
1190
|
const selection = yield* runTransaction("select maintenance lane", () =>
|
|
@@ -900,11 +1202,12 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
900
1202
|
),
|
|
901
1203
|
);
|
|
902
1204
|
|
|
903
|
-
const next =
|
|
904
|
-
runnable.find(
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
1205
|
+
const next = dispatch
|
|
1206
|
+
? (runnable.find(
|
|
1207
|
+
(threadId) =>
|
|
1208
|
+
state.lastServedThreadId === undefined || threadId > state.lastServedThreadId,
|
|
1209
|
+
) ?? runnable[0])
|
|
1210
|
+
: undefined;
|
|
908
1211
|
|
|
909
1212
|
if (next !== undefined) {
|
|
910
1213
|
await transaction.put(
|
|
@@ -915,22 +1218,54 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
915
1218
|
);
|
|
916
1219
|
}
|
|
917
1220
|
|
|
918
|
-
|
|
1221
|
+
const backlog = recovery.started
|
|
1222
|
+
? []
|
|
1223
|
+
: [...heads.keys()].filter((threadId) => {
|
|
1224
|
+
const fault = recoveryFaults.get(threadId);
|
|
1225
|
+
|
|
1226
|
+
return (
|
|
1227
|
+
!eligible.includes(threadId) &&
|
|
1228
|
+
(fault === undefined || fault.retryAt <= selectionTime)
|
|
1229
|
+
);
|
|
1230
|
+
});
|
|
1231
|
+
|
|
1232
|
+
// One finite wave, one Thread at a time, with the runtime's per-Thread deadline.
|
|
1233
|
+
// This cursor ensures an event deadline/eviction cannot always restart at the front.
|
|
1234
|
+
const after = backlog.filter(
|
|
1235
|
+
(threadId) =>
|
|
1236
|
+
state.lastRecoveredThreadId === undefined || threadId > state.lastRecoveredThreadId,
|
|
1237
|
+
);
|
|
1238
|
+
|
|
1239
|
+
const before = backlog.filter(
|
|
1240
|
+
(threadId) =>
|
|
1241
|
+
state.lastRecoveredThreadId !== undefined &&
|
|
1242
|
+
threadId <= state.lastRecoveredThreadId,
|
|
1243
|
+
);
|
|
1244
|
+
|
|
1245
|
+
return { selected: next, retries, backlog: [...after, ...before] };
|
|
919
1246
|
}),
|
|
920
1247
|
);
|
|
921
1248
|
|
|
922
1249
|
yield* failpoint.hit("maintenance:select:after");
|
|
1250
|
+
if (!recovery.started) {
|
|
1251
|
+
recovery.started = true;
|
|
1252
|
+
recovery.needsCheckpoint = selection.backlog.length > 0;
|
|
1253
|
+
for (const threadId of selection.backlog) recovery.pending.add(threadId);
|
|
1254
|
+
yield* Deferred.succeed(recovery.queue, selection.backlog);
|
|
1255
|
+
}
|
|
923
1256
|
|
|
924
1257
|
const selected =
|
|
925
1258
|
selection.selected === undefined ? undefined : heads.get(selection.selected);
|
|
926
1259
|
|
|
1260
|
+
if (selected !== undefined) yield* recoverThread(selected.threadId, recovery);
|
|
1261
|
+
|
|
927
1262
|
let retries = selection.retries;
|
|
928
1263
|
let bindingFailure: DurableBindingFailure | undefined;
|
|
929
1264
|
|
|
930
1265
|
// One runnable FIFO head per native opportunity. An absent agent waits for a deployment,
|
|
931
1266
|
// including for children; other local lanes and host deliveries remain independently due.
|
|
932
1267
|
const settlement =
|
|
933
|
-
selected === undefined
|
|
1268
|
+
selected === undefined || recoveryFaults.has(selected.threadId)
|
|
934
1269
|
? Option.none()
|
|
935
1270
|
: yield* runtime.processThreadHead(selected.threadId, { yieldAfter }).pipe(
|
|
936
1271
|
Effect.catchTag("BindingUnavailable", (failure) => {
|
|
@@ -999,11 +1334,10 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
999
1334
|
const waitingHeads = new Map<ThreadId, boolean>();
|
|
1000
1335
|
|
|
1001
1336
|
const autonomous = remaining.some((snapshot) => {
|
|
1002
|
-
if (snapshot.state === "unknown" &&
|
|
1337
|
+
if (snapshot.state === "unknown" && waiting(snapshot)) return false;
|
|
1003
1338
|
const headWaiting = waitingHeads.get(snapshot.threadId);
|
|
1004
1339
|
|
|
1005
|
-
if (headWaiting === undefined)
|
|
1006
|
-
waitingHeads.set(snapshot.threadId, stableExternalWait(snapshot, reports));
|
|
1340
|
+
if (headWaiting === undefined) waitingHeads.set(snapshot.threadId, waiting(snapshot));
|
|
1007
1341
|
// FIFO followers cannot execute through a stable external wait. Only plain queued
|
|
1008
1342
|
// input is dormant here; admission repairs and accepted aborts still need a pass.
|
|
1009
1343
|
if (
|
|
@@ -1013,26 +1347,28 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
1013
1347
|
)
|
|
1014
1348
|
return false;
|
|
1015
1349
|
|
|
1016
|
-
return !
|
|
1350
|
+
return !waiting(snapshot);
|
|
1017
1351
|
});
|
|
1018
1352
|
|
|
1019
|
-
const progressed =
|
|
1020
|
-
Option.isSome(settlement) ||
|
|
1021
|
-
recovered.some((report) => report.disposition === "repaired");
|
|
1353
|
+
const progressed = Option.isSome(settlement) || recovery.repaired;
|
|
1022
1354
|
|
|
1023
1355
|
const now = yield* Clock.currentTimeMillis;
|
|
1024
1356
|
const ordinaryDelay = autonomous ? yield* rearmDelay(progressed, started.stalls) : 0;
|
|
1025
1357
|
|
|
1026
|
-
const nextEligible =
|
|
1027
|
-
(
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
1358
|
+
const nextEligible = [...recoveryFaults.values()]
|
|
1359
|
+
.map((fault) => fault.retryAt)
|
|
1360
|
+
.concat(
|
|
1361
|
+
eligible.map(
|
|
1362
|
+
(threadId) =>
|
|
1363
|
+
retries.find((retry) => retry.submissionId === heads.get(threadId)?.submissionId)
|
|
1364
|
+
?.notBefore ?? now,
|
|
1365
|
+
),
|
|
1366
|
+
);
|
|
1031
1367
|
|
|
1032
|
-
const
|
|
1368
|
+
const retryDelay =
|
|
1033
1369
|
nextEligible.length === 0 ? 0 : Math.max(0, Math.min(...nextEligible) - now);
|
|
1034
1370
|
|
|
1035
|
-
const delay = Math.max(ordinaryDelay,
|
|
1371
|
+
const delay = Math.max(ordinaryDelay, retryDelay);
|
|
1036
1372
|
|
|
1037
1373
|
// Checkpoint native progress without changing the physical alarm. Auxiliary
|
|
1038
1374
|
// delivery remains live; later mutations still advance the shared generation.
|
|
@@ -1044,11 +1380,14 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
1044
1380
|
const { state } = await readMaintenanceState(transaction);
|
|
1045
1381
|
|
|
1046
1382
|
const processed =
|
|
1047
|
-
autonomous ||
|
|
1383
|
+
autonomous ||
|
|
1384
|
+
observation.activeAtStart > 0 ||
|
|
1385
|
+
started.activeAtStart > 0 ||
|
|
1386
|
+
active > 0
|
|
1048
1387
|
? state.processed
|
|
1049
|
-
: state.processed >
|
|
1388
|
+
: state.processed > observation.generation
|
|
1050
1389
|
? state.processed
|
|
1051
|
-
:
|
|
1390
|
+
: observation.generation;
|
|
1052
1391
|
|
|
1053
1392
|
const next = ThreadMaintenanceState.make({
|
|
1054
1393
|
...Struct.omit(state, ["retry"]),
|
|
@@ -1087,7 +1426,6 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
1087
1426
|
|
|
1088
1427
|
return {
|
|
1089
1428
|
phase: "actionable",
|
|
1090
|
-
recovered: recovered.length,
|
|
1091
1429
|
settled: Option.isSome(settlement) ? 1 : 0,
|
|
1092
1430
|
nonterminal: remaining.length,
|
|
1093
1431
|
nextAttemptAt,
|
|
@@ -1106,7 +1444,9 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
1106
1444
|
Effect.catch(() => Effect.never),
|
|
1107
1445
|
);
|
|
1108
1446
|
|
|
1109
|
-
|
|
1447
|
+
// A wake may observe temporary backoff while this event's recovery is still running.
|
|
1448
|
+
// Its completion must retain the original actionable observation for acknowledgement.
|
|
1449
|
+
const started = yield* beginNative(observed);
|
|
1110
1450
|
|
|
1111
1451
|
// This scope owns auxiliary dispatch and listeners, independently of native progress.
|
|
1112
1452
|
// Close it before final alarm rearming, including on failure or event interruption.
|
|
@@ -1116,11 +1456,127 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
1116
1456
|
|
|
1117
1457
|
const dispatchClosed = yield* Deferred.make<void>();
|
|
1118
1458
|
const stopDispatch = Deferred.await(dispatchClosed);
|
|
1459
|
+
// Coalesce local scan hints while a lane is busy; one hint requests a current read.
|
|
1460
|
+
const checks = yield* PubSub.sliding<void>(1);
|
|
1461
|
+
|
|
1462
|
+
yield* Effect.addFinalizer(() => PubSub.shutdown(checks));
|
|
1463
|
+
let dispatchOpen = true;
|
|
1464
|
+
let activityChanged = yield* Deferred.make<void>();
|
|
1465
|
+
const signalActivity = Effect.suspend(() => Deferred.succeed(activityChanged, undefined));
|
|
1466
|
+
|
|
1467
|
+
const closeDispatch = Effect.sync(() => {
|
|
1468
|
+
dispatchOpen = false;
|
|
1469
|
+
}).pipe(Effect.andThen(Deferred.succeed(dispatchClosed, undefined)));
|
|
1470
|
+
|
|
1471
|
+
const makeActivity = Effect.fnUntraced(function* (allowance?: number) {
|
|
1472
|
+
const initialized = yield* Deferred.make<void>();
|
|
1473
|
+
let active = 0;
|
|
1474
|
+
let failed = false;
|
|
1475
|
+
|
|
1476
|
+
const ready = Deferred.succeed(initialized, undefined).pipe(
|
|
1477
|
+
Effect.andThen(signalActivity),
|
|
1478
|
+
Effect.asVoid,
|
|
1479
|
+
Effect.uninterruptible,
|
|
1480
|
+
);
|
|
1481
|
+
|
|
1482
|
+
const activity: ThreadMaintenanceActivity["Service"] = {
|
|
1483
|
+
ready,
|
|
1484
|
+
subscribeChanges: PubSub.subscribe(checks).pipe(Effect.map(PubSub.take)),
|
|
1485
|
+
run: (wave) =>
|
|
1486
|
+
Effect.acquireUseRelease(
|
|
1487
|
+
Effect.sync(() => {
|
|
1488
|
+
if (!dispatchOpen) return false;
|
|
1489
|
+
active++;
|
|
1490
|
+
|
|
1491
|
+
return true;
|
|
1492
|
+
}),
|
|
1493
|
+
(admitted) =>
|
|
1494
|
+
!admitted
|
|
1495
|
+
? Effect.void
|
|
1496
|
+
: (allowance === undefined
|
|
1497
|
+
? wave
|
|
1498
|
+
: wave.pipe(
|
|
1499
|
+
Effect.timeoutOrElse({
|
|
1500
|
+
duration: allowance,
|
|
1501
|
+
orElse: () =>
|
|
1502
|
+
DurableAlarmError.make({
|
|
1503
|
+
operation: "host dispatch allowance",
|
|
1504
|
+
message:
|
|
1505
|
+
"The admitted host wave exceeded its allowance; durable work remains pending",
|
|
1506
|
+
}),
|
|
1507
|
+
}),
|
|
1508
|
+
)
|
|
1509
|
+
).pipe(
|
|
1510
|
+
Effect.onExit((exit) =>
|
|
1511
|
+
Effect.sync(() => {
|
|
1512
|
+
if (Exit.isFailure(exit)) failed = true;
|
|
1513
|
+
}),
|
|
1514
|
+
),
|
|
1515
|
+
),
|
|
1516
|
+
(admitted) =>
|
|
1517
|
+
Effect.sync(() => {
|
|
1518
|
+
if (admitted) active--;
|
|
1519
|
+
}).pipe(Effect.andThen(signalActivity)),
|
|
1520
|
+
),
|
|
1521
|
+
};
|
|
1522
|
+
|
|
1523
|
+
return {
|
|
1524
|
+
activity,
|
|
1525
|
+
initialized: Deferred.await(initialized),
|
|
1526
|
+
quiet: () => active === 0 && Deferred.isDoneUnsafe(initialized),
|
|
1527
|
+
failed: () => failed,
|
|
1528
|
+
};
|
|
1529
|
+
});
|
|
1530
|
+
|
|
1531
|
+
const messageActivity = yield* makeActivity();
|
|
1532
|
+
const hostActivity = yield* makeActivity(host.dispatchTimeoutMillis);
|
|
1533
|
+
|
|
1534
|
+
const recovery: NativeRecovery = {
|
|
1535
|
+
queue: yield* Deferred.make<ReadonlyArray<ThreadId>>(),
|
|
1536
|
+
pending: new Set(),
|
|
1537
|
+
loaded: new Set(),
|
|
1538
|
+
reports: new Map(),
|
|
1539
|
+
faults: new Map(),
|
|
1540
|
+
started: false,
|
|
1541
|
+
needsCheckpoint: false,
|
|
1542
|
+
recovered: 0,
|
|
1543
|
+
repaired: false,
|
|
1544
|
+
};
|
|
1545
|
+
|
|
1546
|
+
const recoveryFiber = yield* Effect.forkIn(
|
|
1547
|
+
Effect.gen(function* () {
|
|
1548
|
+
for (const threadId of yield* Deferred.await(recovery.queue)) {
|
|
1549
|
+
yield* failpoint.hit("maintenance:select:before");
|
|
1550
|
+
yield* runTransaction("select old recovery lane", () =>
|
|
1551
|
+
ctx.storage.transaction(async (transaction) => {
|
|
1552
|
+
const { state } = await readMaintenanceState(transaction);
|
|
1553
|
+
|
|
1554
|
+
await transaction.put(
|
|
1555
|
+
MAINTENANCE_STATE_KEY,
|
|
1556
|
+
encodeMaintenanceState(
|
|
1557
|
+
ThreadMaintenanceState.make({ ...state, lastRecoveredThreadId: threadId }),
|
|
1558
|
+
),
|
|
1559
|
+
);
|
|
1560
|
+
}),
|
|
1561
|
+
);
|
|
1562
|
+
yield* failpoint.hit("maintenance:select:after");
|
|
1563
|
+
yield* recoverThread(threadId, recovery);
|
|
1564
|
+
recovery.pending.delete(threadId);
|
|
1565
|
+
yield* wakes.notify(threadId);
|
|
1566
|
+
}
|
|
1567
|
+
}),
|
|
1568
|
+
auxiliaryScope,
|
|
1569
|
+
);
|
|
1119
1570
|
|
|
1120
1571
|
// Fork setup too: an ordinary auxiliary setup failure is reported after native work,
|
|
1121
1572
|
// rather than gating its opportunity. Event interruption still closes every fiber.
|
|
1122
1573
|
const deliveryFiber = yield* Effect.forkIn(
|
|
1123
|
-
Scope.provide(auxiliaryScope)(
|
|
1574
|
+
Scope.provide(auxiliaryScope)(
|
|
1575
|
+
messages.drainUntil(stopDispatch, dispatchUntil).pipe(
|
|
1576
|
+
Effect.provideService(ThreadMaintenanceActivity, messageActivity.activity),
|
|
1577
|
+
Effect.onExit(() => messageActivity.activity.ready),
|
|
1578
|
+
),
|
|
1579
|
+
),
|
|
1124
1580
|
auxiliaryScope,
|
|
1125
1581
|
);
|
|
1126
1582
|
|
|
@@ -1137,8 +1593,29 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
1137
1593
|
}),
|
|
1138
1594
|
),
|
|
1139
1595
|
);
|
|
1140
|
-
|
|
1141
|
-
|
|
1596
|
+
|
|
1597
|
+
// Initial setup must also be finite. Once it is accounted for, each actual
|
|
1598
|
+
// wave has its own unchanged allowance; passive listeners have no timer.
|
|
1599
|
+
const initialization = hostActivity.initialized.pipe(
|
|
1600
|
+
Effect.timeoutOrElse({
|
|
1601
|
+
duration: host.dispatchTimeoutMillis,
|
|
1602
|
+
orElse: () =>
|
|
1603
|
+
DurableAlarmError.make({
|
|
1604
|
+
operation: "host initialization",
|
|
1605
|
+
message:
|
|
1606
|
+
"The host did not account for initial maintenance before its allowance; durable work remains pending",
|
|
1607
|
+
}),
|
|
1608
|
+
}),
|
|
1609
|
+
Effect.andThen(Effect.never),
|
|
1610
|
+
);
|
|
1611
|
+
|
|
1612
|
+
yield* Effect.raceFirst(
|
|
1613
|
+
host
|
|
1614
|
+
.drainUntil(stopDispatch, dispatchUntil)
|
|
1615
|
+
.pipe(Effect.provideService(ThreadMaintenanceActivity, hostActivity.activity)),
|
|
1616
|
+
initialization,
|
|
1617
|
+
);
|
|
1618
|
+
}).pipe(Effect.onExit(() => hostActivity.activity.ready)),
|
|
1142
1619
|
),
|
|
1143
1620
|
auxiliaryScope,
|
|
1144
1621
|
);
|
|
@@ -1148,45 +1625,99 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
1148
1625
|
drainDue.pipe(
|
|
1149
1626
|
Effect.provideService(ThreadProjectionMaintenance, projection),
|
|
1150
1627
|
Effect.timeoutOption(config.projectionDispatchTimeoutMillis),
|
|
1628
|
+
Effect.onExit(() => signalActivity),
|
|
1151
1629
|
),
|
|
1152
1630
|
auxiliaryScope,
|
|
1153
1631
|
);
|
|
1154
1632
|
|
|
1155
|
-
let result = yield* advance(started, yieldAfter, observed);
|
|
1633
|
+
let result = yield* advance(started, yieldAfter, observed, recovery);
|
|
1634
|
+
|
|
1635
|
+
// This event owns one finite old-recovery wave, including an empty caught-up wave.
|
|
1636
|
+
recovery.started = true;
|
|
1637
|
+
yield* Deferred.succeed(recovery.queue, []);
|
|
1156
1638
|
let phase = result.phase;
|
|
1157
|
-
let recovered = result.recovered;
|
|
1158
1639
|
let settled = result.settled;
|
|
1159
1640
|
|
|
1160
1641
|
observed.nativeOnly = false;
|
|
1161
1642
|
|
|
1162
|
-
//
|
|
1163
|
-
//
|
|
1164
|
-
|
|
1165
|
-
|
|
1643
|
+
// Pump termination and quiescence are distinct: listeners await closure, while their
|
|
1644
|
+
// finite waves report activity. A held sibling never retires another lane's controls.
|
|
1645
|
+
const hostJoin = yield* Effect.forkIn(
|
|
1646
|
+
Effect.gen(function* () {
|
|
1647
|
+
yield* stopDispatch;
|
|
1166
1648
|
|
|
1167
|
-
|
|
1168
|
-
|
|
1169
|
-
|
|
1170
|
-
|
|
1649
|
+
const remaining = Math.max(
|
|
1650
|
+
1,
|
|
1651
|
+
DateTime.toEpochMillis(dispatchUntil) - (yield* Clock.currentTimeMillis),
|
|
1652
|
+
);
|
|
1171
1653
|
|
|
1172
|
-
|
|
1173
|
-
|
|
1174
|
-
|
|
1175
|
-
|
|
1176
|
-
|
|
1177
|
-
|
|
1178
|
-
),
|
|
1654
|
+
const outcome = yield* Fiber.join(hostFiber).pipe(
|
|
1655
|
+
Effect.timeoutOption(Math.min(host.dispatchTimeoutMillis, remaining)),
|
|
1656
|
+
);
|
|
1657
|
+
|
|
1658
|
+
yield* Effect.annotateCurrentSpan({ "host.timedOut": Option.isNone(outcome) });
|
|
1659
|
+
}),
|
|
1179
1660
|
auxiliaryScope,
|
|
1180
1661
|
);
|
|
1181
1662
|
|
|
1182
|
-
const retired = yield* Effect.forkChild(
|
|
1663
|
+
const retired = yield* Effect.forkChild(
|
|
1664
|
+
Fiber.joinAll([deliveryFiber, hostJoin, backfill, recoveryFiber]),
|
|
1665
|
+
);
|
|
1666
|
+
|
|
1667
|
+
while (true) {
|
|
1668
|
+
const recoveryFinished = recoveryFiber.pollUnsafe() !== undefined;
|
|
1669
|
+
const retiredExit = retired.pollUnsafe();
|
|
1670
|
+
|
|
1671
|
+
if (recoveryFinished) {
|
|
1672
|
+
if (recovery.needsCheckpoint) {
|
|
1673
|
+
yield* Fiber.join(recoveryFiber);
|
|
1674
|
+
// Fold the original recovery observation into its checkpoint before closing.
|
|
1675
|
+
result = yield* advance(started, yieldAfter, observed, recovery, settled === 0);
|
|
1676
|
+
if (result.phase === "actionable") phase = "actionable";
|
|
1677
|
+
settled += result.settled;
|
|
1678
|
+
recovery.needsCheckpoint = false;
|
|
1679
|
+
yield* PubSub.publish(checks, undefined);
|
|
1680
|
+
}
|
|
1681
|
+
}
|
|
1682
|
+
if (retiredExit !== undefined && Exit.isFailure(retiredExit)) break;
|
|
1183
1683
|
|
|
1184
|
-
|
|
1185
|
-
|
|
1186
|
-
|
|
1187
|
-
|
|
1684
|
+
if (
|
|
1685
|
+
recoveryFinished &&
|
|
1686
|
+
backfill.pollUnsafe() !== undefined &&
|
|
1687
|
+
messageActivity.quiet() &&
|
|
1688
|
+
hostActivity.quiet()
|
|
1689
|
+
) {
|
|
1690
|
+
const now = yield* Clock.currentTimeMillis;
|
|
1691
|
+
const messageDeadline = yield* messages.pendingDeadline;
|
|
1692
|
+
const hostDeadline = yield* host.pendingDeadline;
|
|
1693
|
+
|
|
1694
|
+
const messageDue =
|
|
1695
|
+
deliveryFiber.pollUnsafe() === undefined &&
|
|
1696
|
+
!messageActivity.failed() &&
|
|
1697
|
+
Option.isSome(messageDeadline) &&
|
|
1698
|
+
messageDeadline.value <= now;
|
|
1699
|
+
|
|
1700
|
+
const hostDue =
|
|
1701
|
+
hostFiber.pollUnsafe() === undefined &&
|
|
1702
|
+
!hostActivity.failed() &&
|
|
1703
|
+
now + host.dispatchTimeoutMillis <= DateTime.toEpochMillis(dispatchUntil) &&
|
|
1704
|
+
Option.isSome(hostDeadline) &&
|
|
1705
|
+
hostDeadline.value <= now;
|
|
1706
|
+
|
|
1707
|
+
// No asynchronous operation separates the activity recheck from closure. A
|
|
1708
|
+
// racing registration either keeps this window open or cannot start its body.
|
|
1709
|
+
const closed = yield* Effect.sync(() => {
|
|
1710
|
+
if (messageDue || hostDue || !messageActivity.quiet() || !hostActivity.quiet())
|
|
1711
|
+
return false;
|
|
1712
|
+
|
|
1713
|
+
dispatchOpen = false;
|
|
1714
|
+
|
|
1715
|
+
return true;
|
|
1716
|
+
});
|
|
1188
1717
|
|
|
1189
|
-
|
|
1718
|
+
if (closed) break;
|
|
1719
|
+
yield* PubSub.publish(checks, undefined);
|
|
1720
|
+
}
|
|
1190
1721
|
const now = yield* Clock.currentTimeMillis;
|
|
1191
1722
|
const until = DateTime.toEpochMillis(yieldAfter);
|
|
1192
1723
|
|
|
@@ -1198,24 +1729,50 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
1198
1729
|
until,
|
|
1199
1730
|
);
|
|
1200
1731
|
|
|
1732
|
+
// A completion racing this iteration stays armed until the loop handles it.
|
|
1733
|
+
const recoveryDone = recoveryFinished ? Effect.never : Fiber.await(recoveryFiber);
|
|
1734
|
+
const retirementDone = retiredExit === undefined ? Fiber.await(retired) : Effect.never;
|
|
1735
|
+
const changed = activityChanged;
|
|
1736
|
+
|
|
1201
1737
|
const ready = yield* Effect.raceFirst(
|
|
1202
|
-
Effect.raceFirst(notified, Effect.sleep(Math.max(0, next - now))).pipe(
|
|
1203
|
-
|
|
1738
|
+
Effect.raceFirst(notified, Effect.sleep(Math.max(0, next - now))).pipe(
|
|
1739
|
+
Effect.as("native" as const),
|
|
1740
|
+
),
|
|
1741
|
+
Effect.raceFirst(
|
|
1742
|
+
recoveryDone.pipe(Effect.as("recovery" as const)),
|
|
1743
|
+
Effect.raceFirst(
|
|
1744
|
+
retirementDone.pipe(Effect.as("retired" as const)),
|
|
1745
|
+
Deferred.await(changed).pipe(Effect.as("activity" as const)),
|
|
1746
|
+
),
|
|
1747
|
+
),
|
|
1204
1748
|
);
|
|
1205
1749
|
|
|
1206
|
-
if (
|
|
1750
|
+
if (ready === "recovery" || ready === "retired") continue;
|
|
1751
|
+
if (ready === "activity") activityChanged = yield* Deferred.make<void>();
|
|
1207
1752
|
if ((yield* Clock.currentTimeMillis) >= until) break;
|
|
1753
|
+
if (ready === "native") yield* PubSub.publish(checks, undefined);
|
|
1208
1754
|
|
|
1209
|
-
|
|
1210
|
-
|
|
1755
|
+
const awakened = yield* beginNative(observed);
|
|
1756
|
+
|
|
1757
|
+
result = yield* advance(awakened, yieldAfter, observed, recovery);
|
|
1211
1758
|
if (result.phase === "actionable") phase = "actionable";
|
|
1212
|
-
recovered += result.recovered;
|
|
1213
1759
|
settled += result.settled;
|
|
1214
1760
|
observed.nativeOnly = false;
|
|
1761
|
+
if (result.settled > 0) yield* PubSub.publish(checks, undefined);
|
|
1215
1762
|
}
|
|
1763
|
+
// The original native yield deadline closes all new waves even if old recovery
|
|
1764
|
+
// is still pending. No recovery or delivery receives a renewed event budget.
|
|
1765
|
+
yield* closeDispatch;
|
|
1216
1766
|
// Preserve driver-owned Claim deadlines and failures, then close every
|
|
1217
1767
|
// listener before the one final alarm decision.
|
|
1218
1768
|
yield* Fiber.join(retired);
|
|
1769
|
+
if (recovery.needsCheckpoint) {
|
|
1770
|
+
// The native yield deadline ended dispatch before old recovery finished. Fold its
|
|
1771
|
+
// control state into acknowledgement without starting an Attempt after retirement.
|
|
1772
|
+
result = yield* advance(started, yieldAfter, observed, recovery, false);
|
|
1773
|
+
if (result.phase === "actionable") phase = "actionable";
|
|
1774
|
+
settled += result.settled;
|
|
1775
|
+
}
|
|
1219
1776
|
yield* Scope.close(auxiliaryScope, Exit.void);
|
|
1220
1777
|
yield* failpoint.hit("maintenance:finish:before");
|
|
1221
1778
|
|
|
@@ -1256,7 +1813,7 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
1256
1813
|
|
|
1257
1814
|
const report = MaintenancePassReport.make({
|
|
1258
1815
|
phase,
|
|
1259
|
-
recovered,
|
|
1816
|
+
recovered: recovery.recovered,
|
|
1260
1817
|
settled,
|
|
1261
1818
|
nonterminal: result.nonterminal,
|
|
1262
1819
|
alarm: disposition,
|
|
@@ -1309,6 +1866,7 @@ export class ThreadMaintenance extends Context.Service<
|
|
|
1309
1866
|
}),
|
|
1310
1867
|
),
|
|
1311
1868
|
ensureAlarm: mutations.withSnapshot(() => ensureAlarm()),
|
|
1869
|
+
recoveryStatus,
|
|
1312
1870
|
withMutation: (body) =>
|
|
1313
1871
|
mutations.withMutation(
|
|
1314
1872
|
body.pipe(
|