@coreplane/switchboard 1.19.3 → 1.200.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. package/dist/assets/config/config.example.yaml +9 -0
  2. package/dist/assets/deploy/cloudflare/package.json +1 -1
  3. package/dist/assets/deploy/cloudflare-resident/package.json +1 -1
  4. package/dist/assets/deploy/cloudflare-resident/worker.ts +1663 -456
  5. package/dist/assets/deploy/cloudflare-resident/wrangler.template.jsonc +12 -0
  6. package/dist/assets/deploy/cloudflare-sandbox/package.json +1 -1
  7. package/dist/assets/package-lock.json +11 -22
  8. package/dist/assets/package.json +1 -1
  9. package/dist/assets/project.json +1 -1
  10. package/dist/assets/source.json +3 -3
  11. package/dist/assets/src/core/schedules.ts +7 -2
  12. package/dist/assets/src/core/trace/attrs.ts +7 -0
  13. package/dist/assets/src/execution/residentDepsStore.ts +16 -0
  14. package/dist/assets/src/execution/residentIncarnation.ts +134 -0
  15. package/dist/assets/src/execution/residentInstanceId.ts +177 -0
  16. package/dist/assets/src/execution/residentStepPlan.ts +128 -0
  17. package/dist/assets/web/dist/.vite/manifest.json +18 -18
  18. package/dist/assets/web/dist/assets/CostsPage-CzutIHUw.js +1 -0
  19. package/dist/assets/web/dist/assets/{ResidentDetailPage-DvQ05AGa.js → ResidentDetailPage-Dw3e9u1u.js} +1 -1
  20. package/dist/assets/web/dist/assets/{ResidentsIndexPage-B3uxKUne.js → ResidentsIndexPage-D8NN4AFS.js} +1 -1
  21. package/dist/assets/web/dist/assets/{RunRoutePage-ty94olNM.js → RunRoutePage-C1TGcPCl.js} +1 -1
  22. package/dist/assets/web/dist/assets/{RunsIndexPage-CM-qxyQm.js → RunsIndexPage-D9iLpfAr.js} +1 -1
  23. package/dist/assets/web/dist/assets/{ScheduledPage-C1psvLD4.js → ScheduledPage-BJCtMp7H.js} +1 -1
  24. package/dist/assets/web/dist/assets/{StatusDot-DuoQnQeU.js → StatusDot-DvV0z_-s.js} +1 -1
  25. package/dist/assets/web/dist/assets/{Tooltip-BfLPyxQy.js → Tooltip-C7yR095Y.js} +1 -1
  26. package/dist/assets/web/dist/assets/{main-Bnbk_Rsg.js → main-DDu3Ig6S.js} +2 -2
  27. package/dist/cli.js +66 -9
  28. package/package.json +1 -1
  29. package/dist/assets/web/dist/assets/CostsPage-CTZcMYYx.js +0 -1
@@ -77,7 +77,7 @@ import {
77
77
  import { AsyncLocalStorage } from "node:async_hooks";
78
78
  import type { DirectoryBackup, SandboxCommand } from "@cloudflare/sandbox";
79
79
  import { createExtensionProcessSandbox } from "@cloudflare/sandbox/extensions";
80
- import { DurableObject } from "cloudflare:workers";
80
+ import { DurableObject, WorkflowEntrypoint, type WorkflowEvent, type WorkflowStep } from "cloudflare:workers";
81
81
  import { BASH_TIMEOUT_MAX_MS, BASH_TIMEOUT_MS, clampBashTimeout } from "../../src/execution/bashTimeout.js";
82
82
  import { selectBindingsToPurge } from "../../src/execution/bindingPurge.js";
83
83
  import { busyAfterKillReason, planForceDetach } from "../../src/execution/residentDetach.js";
@@ -141,10 +141,47 @@ import {
141
141
  restoreArchivePath,
142
142
  withTimeout,
143
143
  type RefreshDisk,
144
+ type RefreshPlan,
144
145
  type RestoreSample,
145
146
  type RefreshFailure,
146
147
  type RefreshOutcome,
147
148
  } from "../../src/execution/residentRefresh.js";
149
+ import {
150
+ REFRESH_STEP_RETRIES,
151
+ lifecycleOf,
152
+ parseLifecycle,
153
+ refreshInstanceId,
154
+ shouldCreateRefreshInstance,
155
+ stepTimeoutMs,
156
+ type RefreshRow,
157
+ type ResidentLifecycle,
158
+ } from "../../src/execution/residentInstanceId.js";
159
+ import {
160
+ MIRROR_MUTEX_KEY,
161
+ REFRESH_CYCLE_LEASE_MS,
162
+ STALE_MIDFLIGHT_MS,
163
+ depsLeaseKey,
164
+ liveInFlight,
165
+ mintIncarnationId,
166
+ inFlightKey,
167
+ inFlightRow,
168
+ releaseMutex,
169
+ takeMutex,
170
+ type InFlightRow,
171
+ type Lease,
172
+ } from "../../src/execution/residentIncarnation.js";
173
+ import {
174
+ LAST_FETCH_KEY,
175
+ planBuild,
176
+ planFetchMirror,
177
+ planInstallDeps,
178
+ planMaterializeDeps,
179
+ planRestore,
180
+ planSnapshot,
181
+ snapshotCommitDecision,
182
+ type FetchRecord,
183
+ type SnapshotStamp,
184
+ } from "../../src/execution/residentStepPlan.js";
148
185
  import {
149
186
  DF_FREE_ARGV,
150
187
  DISK_FULL_FREE_KIB,
@@ -204,6 +241,8 @@ import {
204
241
  depsEntryPath,
205
242
  depsInstallSemaphoreSize,
206
243
  depsScratchCloneArgv,
244
+ depsAttemptOfScratchPath,
245
+ depsAttemptPaths,
207
246
  depsScratchPath,
208
247
  depsStagingPath,
209
248
  depsHardenScript,
@@ -257,10 +296,21 @@ function refusalOutcome(err: ThreadErr): string {
257
296
  return err.needs ? `needs_${err.needs}` : "error";
258
297
  }
259
298
 
299
+ /** What a refresh instance is created with: the resident it runs for. Every
300
+ * other input is read from the resident's rows at each step, never carried. */
301
+ interface RefreshInstanceParams {
302
+ resource: string;
303
+ }
304
+
260
305
  interface Env {
261
306
  RESIDENT: DurableObjectNamespace<ResidentDO>;
262
307
  REGISTRY: DurableObjectNamespace<ResidentRegistryDO>;
263
308
  BACKUP_BUCKET: R2Bucket;
309
+ /** The refresh cycle as a Workflow instance (docs/reference/specs/resident-repos.md
310
+ * item 7): `ResidentRefresh` below. The watchdog cron creates one per
311
+ * resident whose row says `lifecycle: workflow`; an `alarm` resident (the
312
+ * default) never has one. */
313
+ RESIDENT_REFRESH: Workflow<RefreshInstanceParams>;
264
314
  // Presigned snapshot transfers (docs/reference/specs/resident-repos.md item 61): with all
265
315
  // four present the container moves archive bytes itself over presigned R2
266
316
  // URLs and the DO stays out of the data path; any one absent → the SDK's
@@ -358,6 +408,19 @@ const KILL_EXIT_WAIT_MS = 10_000;
358
408
  * Same class as the other network budgets (observed live transfers run
359
409
  * seconds, recorded in `lastRestore.ms`). */
360
410
  const R2_TRANSFER_TIMEOUT_MS = 5 * 60_000;
411
+ /** The mirror-mutex lease for a section that names no step budget of its own
412
+ * (attach's clone section, a sweep's eviction, the wake and reclaim fetches):
413
+ * a holder of the current incarnation still holding past this has hung, the
414
+ * same bound the watchdog puts on a mid-flight state. The engine steps pass
415
+ * their exact budgets instead. */
416
+ const MIRROR_LEASE_DEFAULT_MS = STALE_MIDFLIGHT_MS;
417
+ /** What the dependency install step runs around the install itself, each
418
+ * bounded: the scratch clone, the seed and its cache swap, the commit (one
419
+ * network budget each) and the harden (the default exec budget). The step's
420
+ * lease is the install budget plus this. */
421
+ const DEPS_STEP_OVERHEAD_MS = 4 * GIT_NETWORK_TIMEOUT_MS + DEFAULT_EXEC_TIMEOUT_MS;
422
+
423
+ const sleep = (ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms));
361
424
 
362
425
  /** On-disk layout inside the resident container (disk is cache, never truth —
363
426
  * DO storage is). Thread worktrees hang off the same mirror; keep these paths stable. */
@@ -453,10 +516,6 @@ const LRU_FLOOR_S = IDLE_AFTER_S;
453
516
  /** Budget for one GitHub REST call in the reclamation pass (pulls lookup per
454
517
  * live non-default binding); a slow API answers "unknown", never blocks the cycle. */
455
518
  const GITHUB_API_TIMEOUT_MS = 10_000;
456
- /** A `refreshing`/`restoring` marker older than this with nothing running is
457
- * an orphan from an interrupted cycle; the watchdog normalizes it. Comfortably
458
- * above the longest legitimate cycle (REFRESH_BUILD_TIMEOUT_MS-scale installs). */
459
- const STALE_MIDFLIGHT_MS = 30 * 60_000;
460
519
  /** A resident degraded with the SAME reason for this many consecutive cycles
461
520
  * is chronically broken (e.g. the default branch's build fails); retrying
462
521
  * every 10 min bills the container 24/7 for nothing. After the streak it may
@@ -493,6 +552,24 @@ const DISK_FULL_REARM_S = 1;
493
552
  * detach and sweep eviction. Surfaced as the live view's `disk`; the attach
494
553
  * admission projects a new tree's cost from its parts. */
495
554
  const DISK_KEY = "resident:disk";
555
+ /** Which scheduler drives this resident's refresh cycle (docs/reference/specs/resident-repos.md
556
+ * item 7): the alarm chain (the default — a row without the key reads `alarm`)
557
+ * or the Workflow instance the watchdog cron creates. Set through the admin
558
+ * `/debug` `lifecycle` op; read by the alarm's entry, the watchdog's re-arm
559
+ * branches and the cron's instance-creation duty, so the two schedulers never
560
+ * both drive a cycle for one resident. */
561
+ const LIFECYCLE_KEY = "resident:lifecycle";
562
+ /** The refresh instance row (item 7): the last instance the cron created for
563
+ * this resident, with the step it last reported and the cycle lease it holds,
564
+ * and the last bucket the cron skipped (a live cycle, a duplicate id). */
565
+ const REFRESH_INSTANCE_KEY = "resident:refreshInstance";
566
+ /** The refresh instance's step budgets: each `step.do` timeout is the DO
567
+ * method's own budget, capped at the engine's 30-minute step ceiling
568
+ * (`stepTimeoutMs`), so a step timeout and a command timeout agree. */
569
+ const REFRESH_FETCH_STEP_BUDGET_MS = RESTORE_MAX_MS + GIT_NETWORK_TIMEOUT_MS; // a wake's restore, then the fetch
570
+ const REFRESH_INSTALL_STEP_BUDGET_MS = REFRESH_INSTALL_TIMEOUT_MS + DEPS_STEP_OVERHEAD_MS; // the install's own lease
571
+ const REFRESH_BUILD_STEP_BUDGET_MS = GIT_NETWORK_TIMEOUT_MS + REFRESH_BUILD_TIMEOUT_MS; // the build's mutex lease
572
+ const REFRESH_SNAPSHOT_STEP_BUDGET_MS = R2_TRANSFER_TIMEOUT_MS + GIT_NETWORK_TIMEOUT_MS; // the archives, then the reclaim pass and the disk sample
496
573
  const DISK_MEASURE_CALLBACK = "onDiskMeasure";
497
574
  const DISK_MEASURE_DELAY_S = 1;
498
575
  /** A `du` over a multi-GB checkout plus every live tree is seconds warm, tens
@@ -1028,6 +1105,61 @@ interface DepsBackupRecord {
1028
1105
  createdAt: string;
1029
1106
  }
1030
1107
 
1108
+ /** What the snapshot step answers: the record already at the stamp, a record
1109
+ * it committed (with the one it replaced), or a step another writer won. */
1110
+ type SnapshotStepResult =
1111
+ | { done: true; record: SnapshotRecord }
1112
+ | { done: false; superseded: false; record: SnapshotRecord; previous: SnapshotRecord | undefined }
1113
+ | { done: false; superseded: true };
1114
+
1115
+ /** The refresh instance row (REFRESH_INSTANCE_KEY, item 7). */
1116
+ interface RefreshInstanceRow {
1117
+ /** The most recent instance the cron created (or a step reported) for this resident. */
1118
+ instance: {
1119
+ id: string;
1120
+ createdAt: string;
1121
+ /** `<step>: <outcome>` of the step the instance last ran; null before its first. */
1122
+ lastStep: string | null;
1123
+ /** The cycle lease the instance holds in the in-flight row, from its fetch step to its end. */
1124
+ holder: string | null;
1125
+ } | null;
1126
+ /** The most recent bucket the cron did not create for: a live cycle, or the engine's duplicate-id refusal. */
1127
+ skipped: { id: string; at: string; why: string } | null;
1128
+ }
1129
+
1130
+ /** What every instance step answers besides its own facts: the resident's
1131
+ * wall clock at the step's start and the commands it ran, so the instance
1132
+ * can graft them under its root the way the bot grafts an attach's. */
1133
+ interface InstanceStepTrace {
1134
+ startedAt: number;
1135
+ trace: ResidentStep[];
1136
+ }
1137
+ /** A step's own verdict: `done` with its facts; `stopped` by a gate that ends
1138
+ * the cycle with nothing to record (idle, a container restart, an offboard);
1139
+ * `failed` by the repository's own doing, already recorded as `degraded`.
1140
+ * A step killed from outside answers none of these — it throws, and the
1141
+ * engine retries it. */
1142
+ type InstanceStepResult<T> =
1143
+ ({ status: "done" } & T) | { status: "stopped"; why: string } | { status: "failed"; reason: string };
1144
+ type InstanceStepAnswer<T> = InstanceStepResult<T> & InstanceStepTrace;
1145
+ /** The fetch step's facts, small by construction: refs, shas, keys, words. */
1146
+ interface RefreshFetchFacts {
1147
+ ref: string;
1148
+ sha: string;
1149
+ factsSha: string;
1150
+ lockfileKey: string;
1151
+ action: RefreshPlan["action"];
1152
+ /** Whether the install step must run: a rebuild whose lockfile key moved, on a repo with an install command. */
1153
+ install: boolean;
1154
+ mintError: string | null;
1155
+ }
1156
+ /** What the cron did about one resident's refresh instance this pass. */
1157
+ interface RefreshInstanceAction {
1158
+ id: string;
1159
+ action: "created" | "duplicate" | "skipped" | "failed";
1160
+ why: string;
1161
+ }
1162
+
1031
1163
  // ---------------------------------------------------------------------------
1032
1164
  // Registry DO (singleton): onboarded set + config, atomic cap enforcement
1033
1165
  // ---------------------------------------------------------------------------
@@ -1385,8 +1517,9 @@ export class ResidentDO extends Sandbox<Env> {
1385
1517
  // because every way the fact can stop being true is observable and clears
1386
1518
  // the memos: a runtime replacement surfaces as RuntimeReplacedError at the
1387
1519
  // ONE exec choke point (`run()`), a deliberate stop/teardown/rebuild calls
1388
- // `clearIncarnationMemos()` at its site, every lifecycle transition
1389
- // (`setResidentState`) clears too, and a sleep cannot race the TTL — the
1520
+ // `swapIncarnation()` at its site (memos AND the incarnation id go), every
1521
+ // lifecycle transition (`setResidentState`) clears the memos alone — the
1522
+ // incarnation survives it — and a sleep cannot race the TTL — the
1390
1523
  // container sleeps only after SLEEP_AFTER (20 min) of idleness, while the
1391
1524
  // hydration memo lives `hydrationMemoTtlMs` (60 s) past the last activity
1392
1525
  // that set it. Storage stays the truth: the memo caches a verdict PROBED from
@@ -1396,6 +1529,29 @@ export class ResidentDO extends Sandbox<Env> {
1396
1529
  private gitSetupDone = false;
1397
1530
  private stageDirsReady = new Set<string>();
1398
1531
 
1532
+ /** The incarnation: one isolate paired with one container runtime
1533
+ * (docs/reference/specs/resident-repos.md item 22). Minted when the object
1534
+ * starts and again ONLY when the runtime is replaced under it
1535
+ * (`swapIncarnation`) — every lease this object writes carries it, and a
1536
+ * lease from another incarnation is a holder whose process context is gone.
1537
+ * A lifecycle transition is not a swap: the cycle that flips the state to
1538
+ * `refreshing` holds a lease of this incarnation and must still hold it
1539
+ * afterwards, so `setResidentState` clears the memos and nothing more. */
1540
+ private incarnation = mintIncarnationId();
1541
+ private leaseSeq = 0;
1542
+
1543
+ private nextHolder(): string {
1544
+ return `${this.incarnation}:${++this.leaseSeq}`;
1545
+ }
1546
+
1547
+ /** The runtime under this object is gone (a replacement seen at the exec
1548
+ * choke point, a deliberate stop/teardown/rebuild, retirement): every lease
1549
+ * this incarnation holds is dead from here on, and so are its memos. */
1550
+ private swapIncarnation(): void {
1551
+ this.incarnation = mintIncarnationId();
1552
+ this.clearIncarnationMemos();
1553
+ }
1554
+
1399
1555
  private clearIncarnationMemos(): void {
1400
1556
  this.hydratedVerdictAt = 0;
1401
1557
  this.gitSetupDone = false;
@@ -1404,18 +1560,25 @@ export class ResidentDO extends Sandbox<Env> {
1404
1560
  this.depsInstallSlots = null;
1405
1561
  }
1406
1562
 
1407
- /** Mirror mutex: a DO yields at every await, so two in-flight
1408
- * requests CAN interleave mid-handler — every mirror mutation (fetch,
1409
- * worktree add/remove) runs under this explicit promise-chain lock. The
1410
- * chain lives in DO memory only; that is sufficient because all mirror
1411
- * work happens through this one DO instance, and a DO restart also drops
1412
- * any in-flight work the lock was guarding. */
1563
+ /** Mirror mutex, in-process half: a DO yields at every await, so two
1564
+ * in-flight requests CAN interleave mid-handler — every mirror mutation
1565
+ * (fetch, worktree add/remove) queues on this promise chain, in arrival
1566
+ * order. The chain is the fast path within one incarnation; the durable
1567
+ * row (MIRROR_MUTEX_KEY) is the truth across incarnations: an isolate swap
1568
+ * drops the chain while the holder's process may keep writing, and the row
1569
+ * is what the next incarnation reads before it touches the tree. */
1413
1570
  private mirrorLockTail: Promise<void> = Promise.resolve();
1414
1571
 
1415
1572
  /** Run `fn` holding the mirror mutex. With waitTimeoutMs > 0, gives up
1416
1573
  * waiting after that long (throws MirrorBusyError) — the queued slot is
1417
- * released so later waiters are not stuck behind a ghost. */
1418
- private async withMirrorLock<T>(fn: () => Promise<T>, waitTimeoutMs = 0): Promise<{ value: T; waitedMs: number }> {
1574
+ * released so later waiters are not stuck behind a ghost. `lease` names
1575
+ * the step and its budget on the row; a section without a budget of its
1576
+ * own gets the default lease. */
1577
+ private async withMirrorLock<T>(
1578
+ fn: () => Promise<T>,
1579
+ waitTimeoutMs = 0,
1580
+ lease: { step: string; budgetMs: number } = { step: "mirror", budgetMs: MIRROR_LEASE_DEFAULT_MS },
1581
+ ): Promise<{ value: T; waitedMs: number }> {
1419
1582
  const prev = this.mirrorLockTail;
1420
1583
  let release!: () => void;
1421
1584
  const slot = new Promise<void>((resolve) => (release = resolve));
@@ -1445,15 +1608,88 @@ export class ResidentDO extends Sandbox<Env> {
1445
1608
  } else {
1446
1609
  await prev.catch(() => {});
1447
1610
  }
1611
+ // Past the chain, the row: taken when free or when its holder is dead.
1612
+ const holder = this.nextHolder();
1613
+ try {
1614
+ await this.takeMirrorRow(holder, lease, waitTimeoutMs, started);
1615
+ } catch (err) {
1616
+ release();
1617
+ throw err;
1618
+ }
1448
1619
  const waitedMs = systemClock() - started;
1449
1620
  this.stepTrace.getStore()?.mutexWait(waitedMs, systemClock());
1450
1621
  try {
1451
1622
  return { value: await fn(), waitedMs };
1452
1623
  } finally {
1453
- release();
1624
+ try {
1625
+ // A failed release must not replace fn's result: the row's expiry is
1626
+ // the backstop, and the next taker takes over a dead holder anyway.
1627
+ await this.releaseLease(MIRROR_MUTEX_KEY, holder).catch((err: unknown) => {
1628
+ console.log(`mirror mutex: release of ${holder} failed (${errMsg(err)}); the lease expires on its own`);
1629
+ });
1630
+ } finally {
1631
+ release();
1632
+ }
1633
+ }
1634
+ }
1635
+
1636
+ /** Write the mirror-mutex row for `holder`, or wait for a live holder of
1637
+ * this incarnation to end. A dead holder — another incarnation, or one past
1638
+ * its budget — is taken over at once (takeMutex); the row only ever names a
1639
+ * live one of THIS incarnation when a release is racing this read, so the
1640
+ * wait is short and bounded by the caller's timeout like the chain wait. */
1641
+ private async takeMirrorRow(
1642
+ holder: string,
1643
+ lease: { step: string; budgetMs: number },
1644
+ waitTimeoutMs: number,
1645
+ started: number,
1646
+ ): Promise<void> {
1647
+ for (;;) {
1648
+ const row = await this.ctx.storage.get<Lease>(MIRROR_MUTEX_KEY);
1649
+ const decision = takeMutex(row, systemClock(), this.incarnation, lease.budgetMs, lease.step, holder);
1650
+ if (decision.action === "take") {
1651
+ if (decision.dead) {
1652
+ console.log(
1653
+ `mirror mutex: ${decision.why} — ${lease.step} takes over from ${decision.dead.step} (${decision.dead.holder})`,
1654
+ );
1655
+ }
1656
+ await this.ctx.storage.put(MIRROR_MUTEX_KEY, decision.row);
1657
+ return;
1658
+ }
1659
+ const left = waitTimeoutMs > 0 ? waitTimeoutMs - (systemClock() - started) : Infinity;
1660
+ if (left <= 0) throw new MirrorBusyError(`mirror-busy: mutex not acquired within ${waitTimeoutMs}ms`);
1661
+ await sleep(Math.min(decision.remainingMs + 1, left, 1_000));
1454
1662
  }
1455
1663
  }
1456
1664
 
1665
+ /** Delete a lease row iff `holder` still owns it (releaseMutex). */
1666
+ private async releaseLease(key: string, holder: string): Promise<void> {
1667
+ const outcome = releaseMutex(await this.ctx.storage.get<Lease>(key), holder);
1668
+ if (outcome.released) await this.ctx.storage.delete(key);
1669
+ }
1670
+
1671
+ /** Lease one of the in-flight facts the watchdog reads: one document per
1672
+ * fact (inFlightKey), a plain put, so a hydration's clear can never race a
1673
+ * refresh's record on a shared row. */
1674
+ private async recordInFlight(kind: keyof InFlightRow, holder: string, budgetMs: number, step: string): Promise<void> {
1675
+ const lease: Lease = { holder, incarnation: this.incarnation, expiresAt: systemClock() + budgetMs, step };
1676
+ await this.ctx.storage.put(inFlightKey(kind), lease);
1677
+ }
1678
+
1679
+ private async clearInFlight(kind: keyof InFlightRow, holder: string): Promise<void> {
1680
+ const lease = await this.ctx.storage.get<Lease>(inFlightKey(kind));
1681
+ if (lease?.holder === holder) await this.ctx.storage.delete(inFlightKey(kind));
1682
+ }
1683
+
1684
+ /** The two in-flight facts as one row, for liveInFlight and /debug. */
1685
+ private async readInFlight(): Promise<InFlightRow> {
1686
+ const [refresh, hydration] = await Promise.all([
1687
+ this.ctx.storage.get<Lease>(inFlightKey("refresh")),
1688
+ this.ctx.storage.get<Lease>(inFlightKey("hydration")),
1689
+ ]);
1690
+ return inFlightRow(refresh, hydration);
1691
+ }
1692
+
1457
1693
  private registry() {
1458
1694
  return this.env.REGISTRY.get(this.env.REGISTRY.idFromName("registry"));
1459
1695
  }
@@ -1491,7 +1727,7 @@ export class ResidentDO extends Sandbox<Env> {
1491
1727
  // takes the throw below. It exists so that if a future SDK vouches "never
1492
1728
  // started" we retry then — and only then — without a change here.
1493
1729
  if (!(err instanceof OperationInterruptedError && err.retryable === true)) {
1494
- this.clearIncarnationMemos(); // the container this incarnation's memos described is gone
1730
+ this.swapIncarnation(); // the container this incarnation's memos described is gone
1495
1731
  throw new RuntimeReplacedError("spawn", err);
1496
1732
  }
1497
1733
  console.log(
@@ -1504,7 +1740,7 @@ export class ResidentDO extends Sandbox<Env> {
1504
1740
  return { stdout: out.stdout, stderr: out.stderr, exitCode: out.exitCode, timedOut: out.timedOut };
1505
1741
  } catch (err) {
1506
1742
  if (isRuntimeReplacement(err)) {
1507
- this.clearIncarnationMemos(); // the container this incarnation's memos described is gone
1743
+ this.swapIncarnation(); // the container this incarnation's memos described is gone
1508
1744
  throw new RuntimeReplacedError("collect", err);
1509
1745
  }
1510
1746
  if (err instanceof ProcessWaitTimeoutError) {
@@ -1681,14 +1917,21 @@ export class ResidentDO extends Sandbox<Env> {
1681
1917
  return (await this.runOk(["sh", "-c", script], "lockfile-key")).trim();
1682
1918
  }
1683
1919
 
1684
- /** True when the disk already holds exactly what the snapshot stamp says. */
1685
- private async diskMatches(sha: string): Promise<boolean> {
1920
+ /** The sha the disk was last materialized to (READY_MARKER), or null when
1921
+ * either tree or the marker is missing — the restore step's fact. */
1922
+ private async readyStamp(): Promise<string | null> {
1686
1923
  const r = await this.run([
1687
1924
  "sh",
1688
1925
  "-c",
1689
1926
  `test -d ${MIRROR_DIR}/objects && test -d ${CHECKOUT_DIR}/.git && cat ${READY_MARKER} 2>/dev/null || echo __absent__`,
1690
1927
  ]);
1691
- return r.exitCode === 0 && r.stdout.trim() === sha;
1928
+ const out = r.stdout.trim();
1929
+ return r.exitCode === 0 && out !== "" && out !== "__absent__" ? out : null;
1930
+ }
1931
+
1932
+ /** True when the disk already holds exactly what the snapshot stamp says. */
1933
+ private async diskMatches(sha: string): Promise<boolean> {
1934
+ return (await this.readyStamp()) === sha;
1692
1935
  }
1693
1936
 
1694
1937
  /** Write the disk markers that a materialized checkout leaves behind (see
@@ -1779,6 +2022,320 @@ export class ResidentDO extends Sandbox<Env> {
1779
2022
  }
1780
2023
  }
1781
2024
 
2025
+ // -- engine steps (docs/reference/specs/resident-repos.md item 22) ---------------
2026
+ //
2027
+ // Each step is one public method a cycle calls: it reads the facts it is
2028
+ // about to change and asks the pure plan (residentStepPlan.ts) whether the
2029
+ // work is done — done issues no command, so a second call with the same
2030
+ // inputs has no effect — then takes its lease, runs its commands under the
2031
+ // step's own budget, writes its result and releases. The alarm chain drives
2032
+ // them today in the order it always did.
2033
+
2034
+ /** Fetch the mirror from origin, once per cycle: the record under
2035
+ * LAST_FETCH_KEY names the cycle, so a repeated call inside the same cycle
2036
+ * answers the sha it already read. */
2037
+ async fetchMirror(input: {
2038
+ ref: string;
2039
+ cycle: string;
2040
+ token: string | null;
2041
+ }): Promise<{ done: boolean; sha: string }> {
2042
+ const last = await this.ctx.storage.get<FetchRecord>(LAST_FETCH_KEY);
2043
+ const plan = planFetchMirror({ ref: input.ref, cycle: input.cycle, last });
2044
+ if (plan.action === "done") return { done: true, sha: plan.sha };
2045
+ // The tip is read under the same lock as the fetch: an attach's own
2046
+ // `fetch --prune` between the two could delete the ref and fail the cycle.
2047
+ const { value: sha } = await this.withMirrorLock(
2048
+ async () => {
2049
+ await this.gitWithCred(
2050
+ input.token,
2051
+ ["-C", MIRROR_DIR, "fetch", "--prune", "origin"],
2052
+ "fetch",
2053
+ GIT_NETWORK_TIMEOUT_MS,
2054
+ );
2055
+ return this.readMirrorSha(input.ref);
2056
+ },
2057
+ 0,
2058
+ { step: "fetch", budgetMs: GIT_NETWORK_TIMEOUT_MS },
2059
+ );
2060
+ await this.ctx.storage.put(LAST_FETCH_KEY, {
2061
+ cycle: input.cycle,
2062
+ ref: input.ref,
2063
+ sha,
2064
+ at: systemClock(),
2065
+ } satisfies FetchRecord);
2066
+ return { done: false, sha };
2067
+ }
2068
+
2069
+ /** The store entry for `key`, complete (item 59 is the primitive under it).
2070
+ * The install's exclusive resource is the key's entry, never the mirror —
2071
+ * installs run in a private scratch tree outside the mirror mutex — so its
2072
+ * lease is per key (depsLeaseKey). A live holder of this incarnation is the
2073
+ * install already running for the key: joined, never duplicated. A dead
2074
+ * holder left its scratch tree behind, possibly with a process still
2075
+ * writing into it: that tree is swept before this attempt starts, the way
2076
+ * every build-user step sweeps the checkout. */
2077
+ async installDeps(input: {
2078
+ key: string;
2079
+ sha: string;
2080
+ installCmd: string;
2081
+ budgetMs: number;
2082
+ seedFromKey?: string;
2083
+ restoreDeadlineMs?: number;
2084
+ }): Promise<{ done: boolean; entry: string }> {
2085
+ const { key } = input;
2086
+ const plan = planInstallDeps({ key, entryComplete: await this.depsEntryComplete(key) });
2087
+ if (plan.action === "done") {
2088
+ await this.run(["touch", depsUsedPath(key)]);
2089
+ return { done: true, entry: depsEntryPath(key) };
2090
+ }
2091
+ const attempt = crypto.randomUUID().slice(0, 8);
2092
+ const leaseKey = depsLeaseKey(key);
2093
+ const holder = this.nextHolder();
2094
+ const leaseMs = Math.max(input.budgetMs, (input.restoreDeadlineMs ?? 0) - systemClock()) + DEPS_STEP_OVERHEAD_MS;
2095
+ for (;;) {
2096
+ const row = await this.ctx.storage.get<Lease>(leaseKey);
2097
+ const decision = takeMutex(
2098
+ row,
2099
+ systemClock(),
2100
+ this.incarnation,
2101
+ leaseMs,
2102
+ "deps-install",
2103
+ holder,
2104
+ depsScratchPath(attempt),
2105
+ );
2106
+ if (decision.action === "wait") {
2107
+ const running = this.depsInFlight.get(key);
2108
+ if (running) return { done: false, entry: await running };
2109
+ // The lease is written before the running install registers itself;
2110
+ // a caller landing in between waits for that, briefly.
2111
+ await sleep(Math.min(decision.remainingMs + 1, 1_000));
2112
+ continue;
2113
+ }
2114
+ if (decision.dead?.tree) {
2115
+ // The dead attempt's scratch tree AND its staging dir: its install ran
2116
+ // in the first, its commit script was moving node_modules into the
2117
+ // second; a process still writing to either is killed, then both go.
2118
+ const deadAttempt = depsAttemptOfScratchPath(decision.dead.tree);
2119
+ const deadPaths = deadAttempt ? depsAttemptPaths(key, deadAttempt) : [decision.dead.tree];
2120
+ console.log(
2121
+ `deps: ${decision.why} — sweeping ${deadPaths.join(" ")} left by ${decision.dead.holder} before installing ${key.slice(0, 8)}`,
2122
+ );
2123
+ for (const dir of deadPaths) {
2124
+ const swept = await this.runOk(killStaleBuildProcessesCommand(BUILD_USER, dir), "deps-install-stale-sweep");
2125
+ if (swept.trim()) console.log(`deps: ${swept.trim()}`);
2126
+ }
2127
+ await this.run(["rm", "-rf", ...deadPaths]).catch(() => {});
2128
+ }
2129
+ await this.ctx.storage.put(leaseKey, decision.row);
2130
+ break;
2131
+ }
2132
+ try {
2133
+ const entry = await this.materializeDeps(key, input.sha, input.installCmd, input.budgetMs, {
2134
+ seedFromKey: input.seedFromKey,
2135
+ restoreDeadlineMs: input.restoreDeadlineMs,
2136
+ attempt,
2137
+ });
2138
+ return { done: false, entry };
2139
+ } finally {
2140
+ await this.releaseLease(leaseKey, holder);
2141
+ }
2142
+ }
2143
+
2144
+ /** Bring the checkout to `sha` and build it, under the mirror mutex. The
2145
+ * disk markers decide (planBuild over the refresh planner): a checkout
2146
+ * whose HEAD, deps and build markers all name the target is done. */
2147
+ async runBuild(input: {
2148
+ sha: string;
2149
+ factsSha: string;
2150
+ lockfileKey: string;
2151
+ buildCmd: string;
2152
+ /** The store entry to link as the checkout's node_modules when the plan installs; null when the command table has no install. */
2153
+ depsEntry: string | null;
2154
+ }): Promise<{ done: boolean; why: string }> {
2155
+ const { sha, lockfileKey } = input;
2156
+ const plan = planBuild({ sha, factsSha: input.factsSha, lockfileKey, disk: await this.readRefreshDisk() });
2157
+ if (plan.action === "done") return { done: true, why: plan.why };
2158
+ // Serialize the CHECKOUT_DIR mutation on the mirror mutex:
2159
+ // materializeThreadDeps reads CHECKOUT_DIR via `cp -al` under the same
2160
+ // lock, so an attach/op dep-copy can never hardlink a half-rebuilt
2161
+ // checkout into a thread tree (torn cache → false ❌ from `repo test`).
2162
+ // No wait timeout, exactly like the fetch lock: the background refresh
2163
+ // queues behind an in-flight attach instead of flipping to degraded on
2164
+ // transient lock contention. Token-free: repo code runs during the build.
2165
+ await this.withMirrorLock(
2166
+ async () => {
2167
+ // Isolation invariant (review 1b): attached, sha-pinned thread
2168
+ // worktrees hold hardlinks to the store entry's FILE inodes, and so
2169
+ // does the checkout. A build that writes THROUGH an existing inode —
2170
+ // many bundlers do (e.g. .next incremental manifests open+truncate
2171
+ // rather than recreate) — would mutate every consumer's pinned
2172
+ // artifacts. The `-x` clean removes the checkout's build output so
2173
+ // the build allocates FRESH inodes; the entry's own files are
2174
+ // owner-read-only (deps-harden), so a write through them fails
2175
+ // loudly instead of silently reaching the store; the tool caches
2176
+ // inside node_modules are the checkout's private copies (item 18).
2177
+ //
2178
+ // Install gate: when the committed lockfile key is unchanged,
2179
+ // node_modules (the view) is excluded from the clean and no deps
2180
+ // work happens; a changed key re-links the view to the new entry —
2181
+ // which is also what drops deps the new lockfile no longer has.
2182
+ await this.buildUserRun(checkoutUpdateCommand(sha, plan.clean), "checkout-update", GIT_NETWORK_TIMEOUT_MS);
2183
+ if (plan.install) {
2184
+ // The old view (a resumed install's keep-deps clean leaves it in
2185
+ // place, item 57) makes way for the new entry's: hardlinks only,
2186
+ // the entry's inodes are untouched.
2187
+ if (input.depsEntry) {
2188
+ await this.runOk(["rm", "-rf", `${CHECKOUT_DIR}/node_modules`], "unlink-deps-view");
2189
+ await this.linkDepsView(`${input.depsEntry}/node_modules`, CHECKOUT_DIR, BUILD_USER);
2190
+ }
2191
+ await this.writeDiskMarkers({ depsKey: lockfileKey });
2192
+ await this.runOk(["rm", "-f", INSTALLING_MARKER], "clear-installing-marker");
2193
+ }
2194
+ await this.buildUserRun(input.buildCmd, "build", REFRESH_BUILD_TIMEOUT_MS);
2195
+ await this.writeDiskMarkers({ builtSha: sha });
2196
+ },
2197
+ 0,
2198
+ { step: "build", budgetMs: GIT_NETWORK_TIMEOUT_MS + REFRESH_BUILD_TIMEOUT_MS },
2199
+ );
2200
+ return { done: false, why: plan.why };
2201
+ }
2202
+
2203
+ /** Archive the mirror and checkout to R2 under `stamp` and record it, with
2204
+ * compare-and-swap on the record read at the start: a record already at
2205
+ * the stamp is done; a record that moved while the archive was taken was
2206
+ * written by someone else and wins — the fresh objects are dropped and the
2207
+ * step answers `superseded`, never a throw. The recorded facts move to the
2208
+ * stamp in the same write, so a wake never sees a half-updated pair. */
2209
+ async snapshot(input: { resource: string; stamp: SnapshotStamp }): Promise<SnapshotStepResult> {
2210
+ const { ref, sha, lockfileHash } = input.stamp;
2211
+ const readAtStart = await this.ctx.storage.get<SnapshotRecord>(SNAPSHOT_KEY);
2212
+ const plan = planSnapshot({ stamp: input.stamp, current: readAtStart });
2213
+ if (plan.action === "done" && readAtStart) return { done: true, record: readAtStart };
2214
+ // Under the mirror mutex: nothing may mutate the checkout while it is archived.
2215
+ const { value: snap } = await this.withMirrorLock(
2216
+ () => this.takeSnapshot(input.resource, ref, sha, lockfileHash),
2217
+ 0,
2218
+ { step: "snapshot", budgetMs: R2_TRANSFER_TIMEOUT_MS },
2219
+ );
2220
+ const stored = await this.ctx.storage.get<SnapshotRecord | RepoFacts>([SNAPSHOT_KEY, FACTS_KEY]);
2221
+ const decision = snapshotCommitDecision({
2222
+ readAtStart,
2223
+ current: stored.get(SNAPSHOT_KEY) as SnapshotRecord | undefined,
2224
+ });
2225
+ if (decision.action === "superseded") {
2226
+ console.log(
2227
+ `snapshot: superseded — a record at ${decision.by?.sha.slice(0, 8) ?? "(none)"} moved under this step`,
2228
+ );
2229
+ await this.deleteBackupObjects([snap.mirror.id, snap.checkout.id]).catch(() => {});
2230
+ return { done: false, superseded: true };
2231
+ }
2232
+ const facts = stored.get(FACTS_KEY) as RepoFacts | undefined;
2233
+ await this.ctx.storage.put({
2234
+ [SNAPSHOT_KEY]: snap,
2235
+ ...(facts
2236
+ ? {
2237
+ [FACTS_KEY]: {
2238
+ ...facts,
2239
+ sha,
2240
+ lockfileHash,
2241
+ lastRefreshAt: new Date(systemClock()).toISOString(),
2242
+ } satisfies RepoFacts,
2243
+ }
2244
+ : {}),
2245
+ });
2246
+ return { done: false, superseded: false, record: snap, previous: readAtStart };
2247
+ }
2248
+
2249
+ /** Bring the disk to the snapshot's stamp: the ready marker naming its sha
2250
+ * is done; otherwise unmount and clean, restore both archives (judged by
2251
+ * their bytes, against the caller's one deadline), verify the restored
2252
+ * mirror against the stamp and hand the checkout to the build user. Runs
2253
+ * inside the hydration lease its caller holds — every mirror-mutex taker
2254
+ * hydrates first, so nothing else touches these trees meanwhile. Failures
2255
+ * leave the resident `down` with the reason and the container stopped, as
2256
+ * the wake path always did. */
2257
+ async restoreCheckout(snap: SnapshotRecord, deadlineMs: number): Promise<{ done: boolean }> {
2258
+ const plan = planRestore({ sha: snap.sha, readyStamp: await this.readyStamp() });
2259
+ if (plan.action === "done") return { done: true };
2260
+ // A restore a previous attempt gave up on may still be writing into these
2261
+ // directories (the SDK call cannot be cancelled): wait for it to
2262
+ // settle before the clean, bounded by the same cap the restores get. A
2263
+ // restore that will not settle even then leaves the disk alone — a named
2264
+ // `down`, not a clean racing a writer.
2265
+ if (this.pendingRestores.size > 0) {
2266
+ try {
2267
+ await withTimeout(
2268
+ Promise.allSettled([...this.pendingRestores]),
2269
+ Math.max(1, deadlineMs - systemClock()),
2270
+ `${this.pendingRestores.size} earlier restore(s) still running`,
2271
+ );
2272
+ } catch (err) {
2273
+ // Same exit as a stalled restore below: the stream is still running and
2274
+ // a rebuild is what follows a `down`, so the container goes with it.
2275
+ this.swapIncarnation(); // deliberate incarnation swap
2276
+ await this.stop().catch((stopErr) => console.log(`restore: stop failed: ${errMsg(stopErr)}`));
2277
+ throw await this.goDown(
2278
+ `r2-restore-failed: ${errMsg(err)} — container stopped so the transfer cannot land on a rebuild`,
2279
+ );
2280
+ }
2281
+ }
2282
+ // A previous incarnation's restore may still be MOUNTED at these paths
2283
+ // (item 61: the SDK's presigned restore mounts) — `rm -rf` on a mount
2284
+ // point is "Device or resource busy". Unmount first, every time.
2285
+ await this.runOk(["sh", "-c", unmountAllRestoresScript()], "unmount-restores");
2286
+ await this.runOk(["rm", "-rf", MIRROR_DIR, CHECKOUT_DIR, ...DISK_MARKERS], "clean-before-restore");
2287
+ try {
2288
+ // The restore pair IS the cold-wake critical path. Sequential on purpose:
2289
+ // the SDK serializes backup operations anyway (one queue), so a
2290
+ // concurrent pair only made the second one's clock run while it waited —
2291
+ // and each is judged by its own bytes (restoreWithProgress), not by
2292
+ // a fixed budget: a slow transfer waits, a stalled one goes down with
2293
+ // the bytes and the idle span named instead of stranding `restoring` for
2294
+ // the watchdog.
2295
+ await this.restoreExtracted(snap.mirror, MIRROR_DIR, "mirror restore", "mirror-restore-extract", deadlineMs);
2296
+ await this.restoreExtracted(
2297
+ snap.checkout,
2298
+ CHECKOUT_DIR,
2299
+ "checkout restore",
2300
+ "checkout-restore-extract",
2301
+ deadlineMs,
2302
+ );
2303
+ } catch (err) {
2304
+ // A stalled or capped restore is STILL STREAMING (the SDK call cannot be
2305
+ // cancelled); `pendingRestores` keeps the next hydrate off its directory,
2306
+ // but a `down` resident's only exit is a REBUILD, and provisioning owns
2307
+ // the same directories. Left running, the restore the wake path gave up
2308
+ // on lands into the checkout the rebuild has just cloned and linked —
2309
+ // tar overwrites in place through the deps store's hardlinks, resetting
2310
+ // every hardened entry file from 444 to 644. Stop
2311
+ // the container on the way down: the disk is ephemeral, the stream dies
2312
+ // with it, and the rebuild starts on an empty one.
2313
+ this.swapIncarnation(); // deliberate incarnation swap
2314
+ await this.stop().catch((stopErr) => console.log(`restore: stop failed: ${errMsg(stopErr)}`));
2315
+ throw await this.goDown(
2316
+ `r2-restore-failed: ${errMsg(err)} — container stopped so the transfer cannot land on a rebuild`,
2317
+ );
2318
+ }
2319
+ await this.ensureGitSetup();
2320
+
2321
+ // Verify the restored disk against the stamp — a snapshot that does not
2322
+ // prove its own {ref, sha, lockfileHash} is refused. Both values
2323
+ // derive from the restored MIRROR (the source of truth the checkout was
2324
+ // built from); the checkout's presence was proven by restoreBackup + the
2325
+ // chown below failing loudly if it is missing.
2326
+ const shaRes = await this.run(["git", "-C", MIRROR_DIR, "rev-parse", "--verify", `refs/heads/${snap.ref}`]);
2327
+ const diskSha = shaRes.exitCode === 0 ? shaRes.stdout.trim() : `unreadable(${tail(shaRes.stderr, 120)})`;
2328
+ const diskLock = await this.lockfileKey(snap.sha).catch((err) => `unreadable(${errMsg(err)})`);
2329
+ if (diskSha !== snap.sha || diskLock !== snap.lockfileHash) {
2330
+ throw await this.goDown(
2331
+ `snapshot-stamp-mismatch: restored disk {sha:${diskSha}, lockfileHash:${diskLock}} != stamp {sha:${snap.sha}, lockfileHash:${snap.lockfileHash}}`,
2332
+ );
2333
+ }
2334
+
2335
+ await this.runOk(["chown", "-R", `${BUILD_USER}:${BUILD_USER}`, CHECKOUT_DIR], "chown");
2336
+ return { done: false };
2337
+ }
2338
+
1782
2339
  /** Delete the R2 objects behind SDK backup handles (backups/<id>/ lives
1783
2340
  * OUTSIDE the resident/<resource>/ prefix, so offboard's prefix sweep
1784
2341
  * cannot reach it — this is the only cleanup path). The ids' prefixes are
@@ -1922,17 +2479,22 @@ export class ResidentDO extends Sandbox<Env> {
1922
2479
  // install budget is at least the refresh's: a provisioning budget below
1923
2480
  // a real install time just fails the onboarding.
1924
2481
  if (record.commands.install) {
1925
- const entry = await this.materializeDeps(
1926
- lockfileHash,
2482
+ const { entry } = await this.installDeps({
2483
+ key: lockfileHash,
1927
2484
  sha,
1928
- record.commands.install,
1929
- Math.max(stepBudget, REFRESH_INSTALL_TIMEOUT_MS),
1930
- );
2485
+ installCmd: record.commands.install,
2486
+ budgetMs: Math.max(stepBudget, REFRESH_INSTALL_TIMEOUT_MS),
2487
+ });
1931
2488
  await this.linkDepsView(`${entry}/node_modules`, CHECKOUT_DIR, BUILD_USER);
1932
2489
  }
1933
2490
  await this.buildUserRun(record.commands.build, "build", stepBudget);
1934
2491
 
1935
- const snap = await this.takeSnapshot(resource, ref, sha, lockfileHash);
2492
+ // Provisioning is the only writer while `onboarding`, so the step cannot
2493
+ // be superseded; a record already at the stamp (a re-fired schedule) is
2494
+ // reused.
2495
+ const snapped = await this.snapshot({ resource, stamp: { ref, sha, lockfileHash } });
2496
+ if (!snapped.done && snapped.superseded) throw new StepError("snapshot", "superseded by a concurrent writer");
2497
+ const snap = snapped.record;
1936
2498
 
1937
2499
  // The deadline may have fired mid-provision (down + slot released);
1938
2500
  // never flip a non-onboarding resident to warm from here.
@@ -1997,12 +2559,18 @@ export class ResidentDO extends Sandbox<Env> {
1997
2559
  // 10-min cadence always outlives the TTL, so a cycle re-probes for real.
1998
2560
  if (this.hydratedVerdictAt !== 0 && systemClock() - this.hydratedVerdictAt < this.hydrationMemoTtlMs) return;
1999
2561
  if (this.hydration) return this.hydration;
2000
- const p = this.doHydrate()
2562
+ // The hydration's lease in the in-flight row (item 22) is what the
2563
+ // watchdog reads: alive for this incarnation until the stale bound, gone
2564
+ // with the isolate that started it.
2565
+ const holder = this.nextHolder();
2566
+ const p = this.recordInFlight("hydration", holder, STALE_MIDFLIGHT_MS, "restore")
2567
+ .then(() => this.doHydrate())
2001
2568
  .then(() => {
2002
2569
  this.hydratedVerdictAt = systemClock();
2003
2570
  })
2004
- .finally(() => {
2571
+ .finally(async () => {
2005
2572
  if (this.hydration === p) this.hydration = null;
2573
+ await this.clearInFlight("hydration", holder);
2006
2574
  });
2007
2575
  this.hydration = p;
2008
2576
  this.hydrationStartedAt = systemClock();
@@ -2044,93 +2612,17 @@ export class ResidentDO extends Sandbox<Env> {
2044
2612
  if (active && (await this.diskMatches(snap.sha))) return;
2045
2613
 
2046
2614
  await this.setResidentState("restoring", "rehydrating");
2047
- if (await this.diskMatches(snap.sha)) {
2615
+ const t0 = systemClock();
2616
+ // One deadline for the whole hydrate: the restore step's wait for earlier
2617
+ // restores, both restores and the deps materialization below judge
2618
+ // against it, so the worst-case `restoring` span is RESTORE_MAX_MS, under
2619
+ // the watchdog's stale-mid-flight window — not three caps in a row.
2620
+ const deadlineMs = systemClock() + RESTORE_MAX_MS;
2621
+ if ((await this.restoreCheckout(snap, deadlineMs)).done) {
2048
2622
  // Raced a container start that already had the right disk.
2049
2623
  await this.setResidentState("warm");
2050
2624
  return;
2051
2625
  }
2052
-
2053
- const t0 = systemClock();
2054
- // A restore a previous attempt gave up on may still be writing into these
2055
- // directories (the SDK call cannot be cancelled): wait for it to
2056
- // settle before the clean, bounded by the same cap the restores get. A
2057
- // restore that will not settle even then leaves the disk alone — a named
2058
- // `down`, not a clean racing a writer.
2059
- // One deadline for the whole hydrate: the wait below and both restores
2060
- // judge against it, so the worst-case `restoring` span is RESTORE_MAX_MS,
2061
- // under the watchdog's stale-mid-flight window — not three caps in a row.
2062
- const deadlineMs = systemClock() + RESTORE_MAX_MS;
2063
- if (this.pendingRestores.size > 0) {
2064
- try {
2065
- await withTimeout(
2066
- Promise.allSettled([...this.pendingRestores]),
2067
- Math.max(1, deadlineMs - systemClock()),
2068
- `${this.pendingRestores.size} earlier restore(s) still running`,
2069
- );
2070
- } catch (err) {
2071
- // Same exit as a stalled restore below: the stream is still running and
2072
- // a rebuild is what follows a `down`, so the container goes with it.
2073
- this.clearIncarnationMemos(); // deliberate incarnation swap
2074
- await this.stop().catch((stopErr) => console.log(`restore: stop failed: ${errMsg(stopErr)}`));
2075
- throw await this.goDown(
2076
- `r2-restore-failed: ${errMsg(err)} — container stopped so the transfer cannot land on a rebuild`,
2077
- );
2078
- }
2079
- }
2080
- // A previous incarnation's restore may still be MOUNTED at these paths
2081
- // (item 61: the SDK's presigned restore mounts) — `rm -rf` on a mount
2082
- // point is "Device or resource busy". Unmount first, every time.
2083
- await this.runOk(["sh", "-c", unmountAllRestoresScript()], "unmount-restores");
2084
- await this.runOk(["rm", "-rf", MIRROR_DIR, CHECKOUT_DIR, ...DISK_MARKERS], "clean-before-restore");
2085
- try {
2086
- // The restore pair IS the cold-wake critical path. Sequential on purpose:
2087
- // the SDK serializes backup operations anyway (one queue), so a
2088
- // concurrent pair only made the second one's clock run while it waited —
2089
- // and each is judged by its own bytes (restoreWithProgress), not by
2090
- // a fixed budget: a slow transfer waits, a stalled one goes down with
2091
- // the bytes and the idle span named instead of stranding `restoring` for
2092
- // the watchdog.
2093
- await this.restoreExtracted(snap.mirror, MIRROR_DIR, "mirror restore", "mirror-restore-extract", deadlineMs);
2094
- await this.restoreExtracted(
2095
- snap.checkout,
2096
- CHECKOUT_DIR,
2097
- "checkout restore",
2098
- "checkout-restore-extract",
2099
- deadlineMs,
2100
- );
2101
- } catch (err) {
2102
- // A stalled or capped restore is STILL STREAMING (the SDK call cannot be
2103
- // cancelled); `pendingRestores` keeps the next hydrate off its directory,
2104
- // but a `down` resident's only exit is a REBUILD, and provisioning owns
2105
- // the same directories. Left running, the restore the wake path gave up
2106
- // on lands into the checkout the rebuild has just cloned and linked —
2107
- // tar overwrites in place through the deps store's hardlinks, resetting
2108
- // every hardened entry file from 444 to 644. Stop
2109
- // the container on the way down: the disk is ephemeral, the stream dies
2110
- // with it, and the rebuild starts on an empty one.
2111
- this.clearIncarnationMemos(); // deliberate incarnation swap
2112
- await this.stop().catch((stopErr) => console.log(`restore: stop failed: ${errMsg(stopErr)}`));
2113
- throw await this.goDown(
2114
- `r2-restore-failed: ${errMsg(err)} — container stopped so the transfer cannot land on a rebuild`,
2115
- );
2116
- }
2117
- await this.ensureGitSetup();
2118
-
2119
- // Verify the restored disk against the stamp — a snapshot that does not
2120
- // prove its own {ref, sha, lockfileHash} is refused. Both values
2121
- // derive from the restored MIRROR (the source of truth the checkout was
2122
- // built from); the checkout's presence was proven by restoreBackup + the
2123
- // chown below failing loudly if it is missing.
2124
- const shaRes = await this.run(["git", "-C", MIRROR_DIR, "rev-parse", "--verify", `refs/heads/${snap.ref}`]);
2125
- const diskSha = shaRes.exitCode === 0 ? shaRes.stdout.trim() : `unreadable(${tail(shaRes.stderr, 120)})`;
2126
- const diskLock = await this.lockfileKey(snap.sha).catch((err) => `unreadable(${errMsg(err)})`);
2127
- if (diskSha !== snap.sha || diskLock !== snap.lockfileHash) {
2128
- throw await this.goDown(
2129
- `snapshot-stamp-mismatch: restored disk {sha:${diskSha}, lockfileHash:${diskLock}} != stamp {sha:${snap.sha}, lockfileHash:${snap.lockfileHash}}`,
2130
- );
2131
- }
2132
-
2133
- await this.runOk(["chown", "-R", `${BUILD_USER}:${BUILD_USER}`, CHECKOUT_DIR], "chown");
2134
2626
  // The snapshot carries the checkout's tree, not the store (item 59): adopt
2135
2627
  // its node_modules as the entry for the stamp's key — a rename plus a
2136
2628
  // hardlink view, seconds — so the first attach on the warm key hits.
@@ -2160,26 +2652,32 @@ export class ResidentDO extends Sandbox<Env> {
2160
2652
  const hasView = (await this.run(["test", "-d", `${CHECKOUT_DIR}/node_modules`])).exitCode === 0;
2161
2653
  if (!record.commands.install) {
2162
2654
  depsLinked = hasView;
2163
- } else if (hasView) {
2164
- depsLinked = true;
2165
2655
  } else {
2166
- const budget = planWakeDepsBudget({
2167
- nowMs: systemClock(),
2168
- deadlineMs,
2169
- installBudgetMs: REFRESH_INSTALL_TIMEOUT_MS,
2656
+ const plan = planMaterializeDeps({
2657
+ key: snap.lockfileHash,
2658
+ viewPresent: hasView,
2659
+ entryComplete: await this.depsEntryComplete(snap.lockfileHash),
2170
2660
  });
2171
- if (budget.action === "skip") throw new Error(`${budget.remainingMs} ms left of the hydrate deadline`);
2172
- const entry = await this.materializeDeps(
2173
- snap.lockfileHash,
2174
- snap.sha,
2175
- record.commands.install,
2176
- budget.installBudgetMs,
2177
- {
2661
+ if (plan.action === "done") {
2662
+ depsLinked = true;
2663
+ } else {
2664
+ console.log(`deps: ${plan.why}`);
2665
+ const budget = planWakeDepsBudget({
2666
+ nowMs: systemClock(),
2667
+ deadlineMs,
2668
+ installBudgetMs: REFRESH_INSTALL_TIMEOUT_MS,
2669
+ });
2670
+ if (budget.action === "skip") throw new Error(`${budget.remainingMs} ms left of the hydrate deadline`);
2671
+ const { entry } = await this.installDeps({
2672
+ key: snap.lockfileHash,
2673
+ sha: snap.sha,
2674
+ installCmd: record.commands.install,
2675
+ budgetMs: budget.installBudgetMs,
2178
2676
  restoreDeadlineMs: deadlineMs,
2179
- },
2180
- );
2181
- await this.linkDepsView(`${entry}/node_modules`, CHECKOUT_DIR, BUILD_USER);
2182
- depsLinked = true;
2677
+ });
2678
+ await this.linkDepsView(`${entry}/node_modules`, CHECKOUT_DIR, BUILD_USER);
2679
+ depsLinked = true;
2680
+ }
2183
2681
  }
2184
2682
  } catch (err) {
2185
2683
  console.log(`deps: no view after restore — the next refresh installs: ${errMsg(err)}`);
@@ -2228,331 +2726,88 @@ export class ResidentDO extends Sandbox<Env> {
2228
2726
 
2229
2727
  private async onRefreshAlarmTraced(payload: string): Promise<void> {
2230
2728
  const resource = payload || ((await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "");
2729
+ // Item 7: a resident on the Workflow lifecycle has no chain. An alarm a
2730
+ // previous flip left armed — or an attach's +1 s pull, or a provisioning's
2731
+ // first arm — runs nothing and re-arms nothing, so the two schedulers never
2732
+ // both drive a cycle for one resident.
2733
+ if ((await this.getLifecycle()) === "workflow") {
2734
+ console.log(`refresh: lifecycle is workflow — the alarm chain runs no cycle for ${resource}`);
2735
+ return;
2736
+ }
2231
2737
  let refreshCounted = false;
2738
+ /** This cycle's lease holder in the in-flight row, once it is counted. */
2739
+ let cycleHolder: string | null = null;
2740
+ /** This firing's identity: what `fetchMirror` records so a repeated call inside the cycle is done. */
2741
+ const cycle = crypto.randomUUID();
2232
2742
  const before = await this.getStatus();
2233
2743
  // down chains stay down (a rebuild is the escape hatch); onboarding is
2234
2744
  // owned by provisioning, which arms the first refresh itself.
2235
2745
  if (before.state === "onboarding" || before.state === "down") return;
2236
2746
  try {
2237
- await this.ensureHydrated();
2238
- const record = await this.registry().getRecord(resource);
2239
- if (!record) return; // offboarded mid-flight: let the chain die quietly
2240
- const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
2241
- if (!facts) throw new StepError("facts", "no repo facts recorded despite hydration");
2242
-
2243
- // Deploy-ordering hazard: `wrangler deploy` swaps the app's image but a
2244
- // RUNNING container keeps the old one, so new Worker code can name pool
2245
- // users the image lacks. Reconcile here (every cycle, cheap) — see
2246
- // reconcileImage — so a rollout self-applies within one refresh.
2247
- if (await this.reconcileImage("refresh")) {
2248
- // Container stopping; it restarts on the new image in seconds. Re-arm
2249
- // SHORT so the resident is re-warmed within a minute instead of
2250
- // sitting on the old cadence for a full 600 s.
2251
- this.rearmOutcome = "image-stale-restart";
2252
- return; // finally re-arms
2253
- }
2254
-
2255
- // Idle sleep: nobody has attached for IDLE_AFTER_S and no live tree is
2256
- // dirty → skip this fetch and park the alarm far out so SLEEP_AFTER can
2257
- // elapse. Staleness is repaid at the next attach (refreshIfStale). A
2258
- // dirty live tree pins the container awake: sleep destroys the disk and
2259
- // uncommitted work is not snapshotted.
2260
- // Only a SETTLED resident may park: a cycle that finds `refreshing`/
2261
- // `restoring` at entry is looking at a marker left by a cycle that died
2262
- // mid-flight (a deploy evicting the DO: stuck `refreshing` + parked →
2263
- // every run falls back cold because the bot's warm-gate probe never
2264
- // sees `warm` again). Run the full cycle instead; it
2265
- // ends warm or degraded, and the next one may park.
2266
- // Decide off a FRESH state read — `before` predates several awaits
2267
- // (hydration, registry, facts, reconcile) — same re-read discipline as
2268
- // every other state decision in this file.
2269
- const entry = await this.getStatus();
2270
- if (entry.state === "degraded" && isDiskFullReason(entry.reason)) {
2271
- // The cycle owns the disk-full verdict (docs/reference/specs/resident-repos.md item 54): re-probe before
2272
- // fetching. Still full → nothing a fetch can do; decide whether the
2273
- // container may be recycled and stop here (a fetch that happened to fit
2274
- // would flip the resident `warm`, the bot would attach, git-setup would
2275
- // fail and flip it back — a flap loop). Space back (a detach or the
2276
- // sweep freed trees) → run the cycle as usual and earn `warm`.
2277
- const free = await this.freeKiB();
2278
- if (free !== null && free < DISK_FULL_FREE_KIB) {
2279
- await this.recoverFromDiskFull(entry.reason, 0);
2280
- return; // finally re-arms: short after a recycle, the cadence otherwise
2281
- }
2282
- }
2283
- let settled = entry.state === "warm";
2284
- if (entry.state === "degraded" && !isNonEvidenceReason(entry.reason)) {
2285
- // Count consecutive cycles that found the same REFRESH-PRODUCED degraded
2286
- // reason (github-unreachable, <step>-failed); a stable streak means
2287
- // retrying is not going to help and parking is the right cost behavior.
2288
- // Any other state resets the streak (below).
2289
- const prev = await this.ctx.storage.get<{ reason: string; count: number }>(DEGRADED_STREAK_KEY);
2290
- const streak =
2291
- prev && prev.reason === entry.reason
2292
- ? { reason: entry.reason, count: prev.count + 1 }
2293
- : { reason: entry.reason, count: 1 };
2294
- await this.ctx.storage.put(DEGRADED_STREAK_KEY, streak);
2295
- settled = streak.count >= DEGRADED_PARK_AFTER_CYCLES;
2296
- } else {
2297
- // Warm, or a degraded stamped by the WATCHDOG (alarm-missed /
2298
- // stale-mid-flight) or by an INTERRUPTED cycle (refresh-interrupted —
2299
- // a deploy killed the step; it says nothing about the repo): the
2300
- // watchdog pulled this cycle to +5s precisely so a refresh RUNS, and the
2301
- // interrupted cycle re-armed short for the same reason.
2302
- // Counting those toward the streak would be self-fulfilling —
2303
- // each cycle that found the reason would park without attempting anything,
2304
- // and after three the resident would sit parked-degraded for 6h at a
2305
- // time. Never settled; streak reset.
2306
- await this.ctx.storage.delete(DEGRADED_STREAK_KEY);
2307
- }
2308
- if (settled && (await this.isIdle())) {
2309
- // isIdle awaited (git status per live tree) — re-read before writing.
2310
- const now = (await this.ctx.storage.get<RepoFacts>(FACTS_KEY)) ?? facts;
2311
- if (!now.idleSince)
2312
- await this.ctx.storage.put(FACTS_KEY, {
2313
- ...now,
2314
- idleSince: new Date(systemClock()).toISOString(),
2315
- } satisfies RepoFacts);
2316
- this.rearmOutcome = "idle";
2317
- return; // finally re-arms at IDLE_REFRESH_INTERVAL_S
2318
- }
2319
- if (facts.idleSince) {
2320
- const now = (await this.ctx.storage.get<RepoFacts>(FACTS_KEY)) ?? facts;
2321
- const { idleSince: _woke, ...awake } = now;
2322
- await this.ctx.storage.put(FACTS_KEY, awake satisfies RepoFacts);
2323
- }
2747
+ const gate = await this.refreshGate(resource);
2748
+ if (!gate.go) return; // finally re-arms on the outcome the gate set
2749
+ const { record, facts } = gate;
2324
2750
  // From here the cycle mutates the mirror/checkout: count it as in flight
2325
- // so an attach-path reconcileImage never stops the container under it.
2751
+ // so an attach-path reconcileImage never stops the container under it,
2752
+ // and lease it in the in-flight row so the watchdog can tell this cycle
2753
+ // from a marker a dead one left behind (item 22).
2326
2754
  this.refreshesInFlight++;
2327
2755
  refreshCounted = true;
2756
+ cycleHolder = this.nextHolder();
2757
+ await this.recordInFlight("refresh", cycleHolder, REFRESH_CYCLE_LEASE_MS, "refresh");
2328
2758
 
2329
- // Token-mint failure is a command-level error — the resident
2330
- // keeps serving the last snapshot and lifecycle state is NOT flipped by
2331
- // it. It is recorded, and the cycle then CONTINUES with an anonymous
2332
- // fetch (exactly what an unconfigured App does): a public repo outside
2333
- // the installation stays fresh, and a private one fails at the fetch
2334
- // below into a visible `degraded(github-unreachable: …)`. Returning here
2335
- // instead would freeze whatever state the resident was in — a public
2336
- // repo the App is not installed on would sit in the watchdog's
2337
- // `degraded(alarm-missed)` forever with an ever-staler mirror, because
2338
- // the App cannot mint for a repo it is not installed on.
2339
- let token: string | null = null;
2340
- // This cycle's mint error, kept so it survives the warm facts write below
2341
- // (which clears errors from PRIOR cycles) and prefixes a fetch failure's
2342
- // reason — the observable for "App configured, repo outside the
2343
- // installation" is a warm-but-anonymous resident with the mint named.
2344
- let mintError: string | undefined;
2345
- if (githubAppConfigured(this.env)) {
2346
- try {
2347
- token = (await mintRepoScopedToken(this.env, resource.slice("repo:".length))).token;
2348
- } catch (err) {
2349
- mintError = `token-mint-failed (command-level, fetching anonymously): ${errMsg(err)}`;
2350
- await this.recordRefreshError(mintError);
2351
- }
2352
- }
2353
-
2354
- await this.setResidentState("refreshing");
2355
- try {
2356
- // Same mirror mutex as attach's fetch/worktree work: the
2357
- // refresh alarm and an in-flight attach serialize instead of racing
2358
- // a prune against a worktree clone.
2359
- await this.withMirrorLock(() =>
2360
- this.gitWithCred(token, ["-C", MIRROR_DIR, "fetch", "--prune", "origin"], "fetch", GIT_NETWORK_TIMEOUT_MS),
2361
- );
2362
- } catch (err) {
2363
- // A private repo whose mint failed lands here (the anonymous fetch is
2364
- // refused): say so, rather than blaming GitHub reachability alone.
2365
- const cause = mintError ? `${mintError}; then ` : "";
2366
- const message = `${cause}${errMsg(err)}`;
2367
- // A full disk fails this step too — the credential file is written
2368
- // here (`ENOSPC` on /workspace/.resident/git-credentials would read as
2369
- // github-unreachable, a SERVICEABLE reason, so every run would attach
2370
- // and die at git-setup). Name the disk instead: not
2371
- // serviceable, and the recovery below can free it.
2372
- const failure = await this.classifyFailure("fetch", message);
2373
- if (failure.diskFull) {
2374
- await this.setResidentState("degraded", failure.reason);
2375
- await this.recoverFromDiskFull(failure.reason, refreshCounted ? 1 : 0);
2376
- return;
2377
- }
2378
- await this.setResidentState("degraded", `github-unreachable: ${message}`);
2379
- return;
2380
- }
2381
-
2382
- const sha = await this.readMirrorSha(facts.defaultRef);
2383
- // Pure function of the commit — computed from the mirror before
2384
- // any checkout work so the planner can compare it to the deps marker.
2385
- const lockfileHash = sha === facts.sha ? facts.lockfileHash : await this.lockfileKey(sha);
2759
+ const fetched = await this.refreshFetch(resource, facts, cycle, 1);
2760
+ if (!fetched.ok) return;
2761
+ const { sha, lockfileHash, mintError, token } = fetched;
2386
2762
  const plan = planRefresh({
2387
2763
  sha,
2388
2764
  factsSha: facts.sha,
2389
2765
  lockfileKey: lockfileHash,
2390
2766
  disk: await this.readRefreshDisk(),
2391
2767
  });
2392
- let snap: SnapshotRecord | null = null;
2393
- let previous: SnapshotRecord | undefined;
2768
+ // Whether this cycle's snapshot step committed (`superseded` means
2769
+ // another writer moved the record, whose facts then stand).
2770
+ let committed = true;
2394
2771
  if (plan.action !== "unchanged") {
2395
2772
  const t0 = systemClock();
2396
2773
  console.log(`refresh: ${facts.sha.slice(0, 8)} → ${sha.slice(0, 8)}: ${plan.action} (${plan.why})`);
2397
- // Serialize the CHECKOUT_DIR mutation on the mirror mutex (FIX 2):
2398
- // materializeThreadDeps reads CHECKOUT_DIR via `cp -al` under the same
2399
- // lock, so an attach/op dep-copy can no longer hardlink a half-rebuilt
2400
- // checkout into a thread tree (torn cache → false ❌ from `repo test`).
2401
- // No wait timeout, exactly like the fetch lock above: the background
2402
- // refresh queues behind an in-flight attach instead of flipping to
2403
- // degraded on transient lock contention.
2404
- // Token-free from here on: repo code runs during install/build.
2405
- //
2406
2774
  // Deps come from the store (item 59): a changed lockfile key is
2407
2775
  // materialized ONCE into `/workspace/deps/<key>` — OUTSIDE the mirror
2408
2776
  // lock, because the install runs in its own scratch clone and touches
2409
2777
  // no consumer's tree (the staging step) — and the checkout's
2410
- // node_modules becomes a hardlink view of that entry. An attach that
2411
- // needs the same key joins this very install instead of starting its
2412
- // own. Checkpoint: the deps marker comes off BEFORE the install so an
2413
- // interruption mid-install can never read as completion.
2778
+ // node_modules becomes a hardlink view of that entry (runBuild). An
2779
+ // attach that needs the same key joins this very install instead of
2780
+ // starting its own. Checkpoint: the deps marker comes off BEFORE the
2781
+ // install so an interruption mid-install can never read as completion.
2414
2782
  let depsEntry: string | null = null;
2415
2783
  if (plan.action === "rebuild") {
2416
- await this.runOk(["rm", "-f", BUILT_MARKER, ...(plan.install ? [DEPS_MARKER] : [])], "clear-markers");
2417
- if (plan.install && record.commands.install) {
2418
- // The installing marker brackets the install (item 57): written
2419
- // before, removed after the deps key lands, so a cycle that ends in
2420
- // between is planned as a resume. The install is seeded from the
2421
- // key the checkout holds now — npm reconciles the delta; a resumed
2422
- // install finds its key already in the store when the last attempt
2423
- // completed, or reconciles from the warm key again when it did not.
2424
- await this.writeDiskMarkers({ installingKey: lockfileHash });
2425
- depsEntry = await this.materializeDeps(
2426
- lockfileHash,
2427
- sha,
2428
- record.commands.install,
2429
- REFRESH_INSTALL_TIMEOUT_MS,
2430
- {
2431
- seedFromKey: facts.lockfileHash,
2432
- },
2433
- );
2434
- }
2784
+ await this.refreshClearMarkers(plan.install);
2785
+ if (plan.install) depsEntry = await this.refreshInstall(record, facts, sha, lockfileHash);
2435
2786
  }
2436
- await this.withMirrorLock(async () => {
2437
- if (plan.action === "rebuild") {
2438
- // Isolation invariant (review 1b): attached, sha-pinned thread
2439
- // worktrees hold hardlinks to the store entry's FILE inodes, and so
2440
- // does the checkout. A build that writes THROUGH an existing inode —
2441
- // many bundlers do (e.g. .next incremental manifests open+truncate
2442
- // rather than recreate) — would mutate every consumer's pinned
2443
- // artifacts. The `-x` clean removes the checkout's build output so
2444
- // the build allocates FRESH inodes; the entry's own files are
2445
- // owner-read-only (deps-harden), so a write through them fails
2446
- // loudly instead of silently reaching the store; the tool caches
2447
- // inside node_modules are the checkout's private copies (item 18).
2448
- //
2449
- // Install gate: when the committed lockfile key is unchanged,
2450
- // node_modules (the view) is excluded from the clean and no deps
2451
- // work happens; a changed key takes the full clean and re-links the
2452
- // view to the new entry — which is also what drops deps the new
2453
- // lockfile no longer has.
2454
- await this.buildUserRun(checkoutUpdateCommand(sha, plan.clean), "checkout-update", GIT_NETWORK_TIMEOUT_MS);
2455
- if (plan.install) {
2456
- // The old view (a resumed install's keep-deps clean leaves it in
2457
- // place, item 57) makes way for the new entry's: hardlinks only,
2458
- // the entry's inodes are untouched.
2459
- if (depsEntry) {
2460
- await this.runOk(["rm", "-rf", `${CHECKOUT_DIR}/node_modules`], "unlink-deps-view");
2461
- await this.linkDepsView(`${depsEntry}/node_modules`, CHECKOUT_DIR, BUILD_USER);
2462
- }
2463
- await this.writeDiskMarkers({ depsKey: lockfileHash });
2464
- await this.runOk(["rm", "-f", INSTALLING_MARKER], "clear-installing-marker");
2465
- }
2466
- await this.buildUserRun(record.commands.build, "build", REFRESH_BUILD_TIMEOUT_MS);
2467
- await this.writeDiskMarkers({ builtSha: sha });
2468
- }
2469
- // `reuse`: the checkout already holds this sha with its deps and
2470
- // build (an interrupted cycle got that far) — only the snapshot,
2471
- // facts and stamp are missing, and they must still move together.
2472
- previous = await this.ctx.storage.get<SnapshotRecord>(SNAPSHOT_KEY);
2473
- snap = await this.takeSnapshot(resource, facts.defaultRef, sha, lockfileHash);
2787
+ // `reuse`: the checkout already holds this sha with its deps and build
2788
+ // (an interrupted cycle got that far) — the build step finds it done and
2789
+ // only the snapshot, facts and stamp are missing; they move together.
2790
+ await this.runBuild({
2791
+ sha,
2792
+ factsSha: facts.sha,
2793
+ lockfileKey: lockfileHash,
2794
+ buildCmd: record.commands.build,
2795
+ depsEntry,
2474
2796
  });
2797
+ committed = await this.refreshSnapshot(resource, { ref: facts.defaultRef, sha, lockfileHash });
2475
2798
  console.log(`refresh: ${sha.slice(0, 8)} ${plan.action} done in ${systemClock() - t0}ms`);
2476
2799
  }
2477
-
2478
- // Facts and snapshot move together so the stamp check never sees a
2479
- // half-updated pair.
2480
- const updatedFacts: RepoFacts = {
2481
- ...facts,
2482
- sha,
2483
- lockfileHash,
2484
- lastRefreshAt: new Date(systemClock()).toISOString(),
2485
- };
2486
- // Clear a PRIOR cycle's error; keep THIS cycle's mint error visible.
2487
- delete updatedFacts.lastRefreshError;
2488
- if (mintError) updatedFacts.lastRefreshError = mintError;
2489
- // A wake cycle cleared idleSince above; `facts` was read at alarm entry and
2490
- // still carries it — never resurrect it here (the dash would show a stale
2491
- // "idle since" and every attach would take the wake-fetch path).
2492
- delete updatedFacts.idleSince;
2493
- if (snap) {
2494
- await this.ctx.storage.put({ [FACTS_KEY]: updatedFacts, [SNAPSHOT_KEY]: snap });
2495
- await this.writeDiskMarkers({ ready: sha });
2496
- if (previous) await this.deleteBackupObjects([previous.mirror.id, previous.checkout.id]).catch(() => {});
2497
- } else {
2498
- await this.ctx.storage.put(FACTS_KEY, updatedFacts);
2499
- }
2500
- await this.setResidentState("warm");
2501
- // Event-triggered reclamation: the prune above already told the
2502
- // mirror which branches died; finished refs give their worktree and
2503
- // pool user back now, not at the idle TTL. Housekeeping, never a
2504
- // lifecycle flip — a failure here is a log line.
2505
- try {
2506
- const gc = await this.reclaimFinishedRefs(resource, facts.defaultRef, token);
2507
- if (gc.reclaimed.length > 0) console.log(`reclaim ${resource}: ${JSON.stringify(gc)}`);
2508
- } catch (err) {
2509
- console.log(`reclaim ${resource}: pass failed: ${errMsg(err)}`);
2510
- }
2511
- // Item 55: the cycle's disk sample — what /residents, `repo list`, the
2512
- // watchdog line and the next attach admission read. Housekeeping too.
2513
- await this.measureDisk().catch((err) => console.log(`disk: measure failed: ${errMsg(err)}`));
2800
+ await this.refreshComplete(resource, facts, { sha, lockfileHash, committed, mintError, token });
2514
2801
  } catch (err) {
2515
2802
  if (err instanceof ResidentDownError) return; // already down with reason; chain stops below
2516
- // A step killed from OUTSIDE (the container replaced under it — an
2517
- // image-changing deploy or a container stop; a Worker-only deploy leaves
2518
- // the container running and interrupts nothing) is
2519
- // `refresh-interrupted`: it is not evidence about the repo — it
2520
- // never counts toward the park streak (the entry gate above) — and the
2521
- // chain re-arms SHORT so the resident is warm again within a minute
2522
- // instead of after the full cadence (an unclassified kill otherwise
2523
- // costs the resident the whole 10-minute cadence, e.g.
2524
- // `degraded(build-failed: exit 143 …)` until the next alarm).
2525
- // Any other failure is the repo's own: `<step>-failed: …` /
2526
- // `refresh-failed: …` as before. Non-StepErrors classify too — an SDK
2527
- // replacement error can surface between steps — with the generic
2528
- // "refresh" step, whose failure reason is the pre-existing
2529
- // `refresh-failed: …` shape.
2530
- // A full disk is a third class: `disk-full: …`, never serviceable,
2531
- // and the one failure the resident can act on itself (recoverFromDiskFull).
2532
- const failure =
2533
- err instanceof StepError
2534
- ? await this.classifyFailure(err.step, err.message)
2535
- : await this.classifyFailure("refresh", errMsg(err));
2803
+ const failure = await this.classifyCycleError(err);
2536
2804
  // Set BEFORE the writes on purpose: if either throws, the finally still
2537
2805
  // re-arms short — the safe direction for an interruption.
2538
2806
  if (failure.interrupted) this.rearmOutcome = "interrupted";
2539
- // The classified reason, always in the log: a StepError logged its own
2540
- // output block above, but a failure between steps (an SDK error, the
2541
- // markers, the snapshot) reached only the state entry — which the next
2542
- // cycle's failure overwrites (an install timeout that starts an
2543
- // incident leaves no trace once the follow-up cycle fails).
2544
- console.log(`refresh: cycle failed — ${failure.reason.slice(0, 400)}`);
2545
- // Record on the facts too: the degraded state write below can be
2546
- // clobbered within seconds by a concurrent attach/exec whose
2547
- // ensureHydrated flips the state to `restoring · rehydrating`, leaving
2548
- // no visible trace of WHY.
2549
- // `lastRefreshError` survives that race and the next completed cycle
2550
- // clears it, same as a mint error.
2551
- await this.recordRefreshError(failure.reason);
2552
- await this.setResidentState("degraded", failure.reason); // last snapshot keeps serving
2553
- if (failure.diskFull) await this.recoverFromDiskFull(failure.reason, refreshCounted ? 1 : 0);
2807
+ await this.refreshFailed(failure, refreshCounted ? 1 : 0);
2554
2808
  } finally {
2555
2809
  if (refreshCounted) this.refreshesInFlight--;
2810
+ if (cycleHolder) await this.clearInFlight("refresh", cycleHolder);
2556
2811
  const state = await this.ctx.storage.get<ResidentState>(STATE_KEY);
2557
2812
  // Consecutive-interruption count: bounds the short re-arm so
2558
2813
  // a step whose output chronically carries the kill signature falls back to
@@ -2571,10 +2826,685 @@ export class ResidentDO extends Sandbox<Env> {
2571
2826
  consecutiveInterrupted,
2572
2827
  });
2573
2828
  this.rearmOutcome = "normal";
2574
- if (state && state !== "down" && state !== "onboarding") await this.armRefresh(resource, interval);
2829
+ // A flip to `workflow` while this cycle ran: the chain ends here (item 7).
2830
+ const chained = (await this.getLifecycle()) === "alarm";
2831
+ if (chained && state && state !== "down" && state !== "onboarding") await this.armRefresh(resource, interval);
2832
+ }
2833
+ }
2834
+
2835
+ // -- the cycle's phases, shared by the alarm and the instance (item 7) -------
2836
+ //
2837
+ // The alarm chain and the Workflow instance run the same cycle in the same
2838
+ // order: the gates, the fetch, the plan, the install, the build, the
2839
+ // snapshot, the completion. Each phase is one method here so neither
2840
+ // scheduler carries a copy; the instance calls them one step at a time
2841
+ // (`refreshInstance*`, below), the alarm in one handler (above).
2842
+
2843
+ /** The cycle's entry gates, in the alarm's order: hydrate; the registry
2844
+ * record (gone → the resident was offboarded mid-flight, nothing to do);
2845
+ * the image reconcile (a stale image stops the container, which restarts
2846
+ * on the current one — `image-stale-restart`); the disk-full re-probe (a
2847
+ * disk still full decides its own recovery and stops the cycle — item 54);
2848
+ * the park streak and the idle gate (`idle`); and, for a cycle that runs,
2849
+ * the end of idle mode. Sets `rearmOutcome` for the alarm's finally; the
2850
+ * instance resets it. */
2851
+ private async refreshGate(
2852
+ resource: string,
2853
+ ): Promise<{ go: false; why: string } | { go: true; record: ResidentRecord; facts: RepoFacts }> {
2854
+ await this.ensureHydrated();
2855
+ const record = await this.registry().getRecord(resource);
2856
+ if (!record) return { go: false, why: "offboarded" }; // offboarded mid-flight: let the chain die quietly
2857
+ const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
2858
+ if (!facts) throw new StepError("facts", "no repo facts recorded despite hydration");
2859
+
2860
+ // Deploy-ordering hazard: `wrangler deploy` swaps the app's image but a
2861
+ // RUNNING container keeps the old one, so new Worker code can name pool
2862
+ // users the image lacks. Reconcile here (every cycle, cheap) — see
2863
+ // reconcileImage — so a rollout self-applies within one refresh.
2864
+ if (await this.reconcileImage("refresh")) {
2865
+ // Container stopping; it restarts on the new image in seconds. Re-arm
2866
+ // SHORT so the resident is re-warmed within a minute instead of
2867
+ // sitting on the old cadence for a full 600 s.
2868
+ this.rearmOutcome = "image-stale-restart";
2869
+ return { go: false, why: "image-stale-restart" };
2870
+ }
2871
+
2872
+ // Idle sleep: nobody has attached for IDLE_AFTER_S and no live tree is
2873
+ // dirty → skip this fetch and park the alarm far out so SLEEP_AFTER can
2874
+ // elapse. Staleness is repaid at the next attach (refreshIfStale). A
2875
+ // dirty live tree pins the container awake: sleep destroys the disk and
2876
+ // uncommitted work is not snapshotted.
2877
+ // Only a SETTLED resident may park: a cycle that finds `refreshing`/
2878
+ // `restoring` at entry is looking at a marker left by a cycle that died
2879
+ // mid-flight (a deploy evicting the DO: stuck `refreshing` + parked →
2880
+ // every run falls back cold because the bot's warm-gate probe never
2881
+ // sees `warm` again). Run the full cycle instead; it
2882
+ // ends warm or degraded, and the next one may park.
2883
+ // Decide off a FRESH state read — the caller's read predates several awaits
2884
+ // (hydration, registry, facts, reconcile) — same re-read discipline as
2885
+ // every other state decision in this file.
2886
+ const entry = await this.getStatus();
2887
+ if (entry.state === "degraded" && isDiskFullReason(entry.reason)) {
2888
+ // The cycle owns the disk-full verdict (docs/reference/specs/resident-repos.md item 54): re-probe before
2889
+ // fetching. Still full → nothing a fetch can do; decide whether the
2890
+ // container may be recycled and stop here (a fetch that happened to fit
2891
+ // would flip the resident `warm`, the bot would attach, git-setup would
2892
+ // fail and flip it back — a flap loop). Space back (a detach or the
2893
+ // sweep freed trees) → run the cycle as usual and earn `warm`.
2894
+ const free = await this.freeKiB();
2895
+ if (free !== null && free < DISK_FULL_FREE_KIB) {
2896
+ await this.recoverFromDiskFull(entry.reason, 0);
2897
+ return { go: false, why: "disk-full" }; // the alarm's finally re-arms: short after a recycle, the cadence otherwise
2898
+ }
2899
+ }
2900
+ let settled = entry.state === "warm";
2901
+ if (entry.state === "degraded" && !isNonEvidenceReason(entry.reason)) {
2902
+ // Count consecutive cycles that found the same REFRESH-PRODUCED degraded
2903
+ // reason (github-unreachable, <step>-failed); a stable streak means
2904
+ // retrying is not going to help and parking is the right cost behavior.
2905
+ // Any other state resets the streak (below).
2906
+ const prev = await this.ctx.storage.get<{ reason: string; count: number }>(DEGRADED_STREAK_KEY);
2907
+ const streak =
2908
+ prev && prev.reason === entry.reason
2909
+ ? { reason: entry.reason, count: prev.count + 1 }
2910
+ : { reason: entry.reason, count: 1 };
2911
+ await this.ctx.storage.put(DEGRADED_STREAK_KEY, streak);
2912
+ settled = streak.count >= DEGRADED_PARK_AFTER_CYCLES;
2913
+ } else {
2914
+ // Warm, or a degraded stamped by the WATCHDOG (alarm-missed /
2915
+ // stale-mid-flight) or by an INTERRUPTED cycle (refresh-interrupted —
2916
+ // a deploy killed the step; it says nothing about the repo): the
2917
+ // watchdog pulled this cycle to +5s precisely so a refresh RUNS, and the
2918
+ // interrupted cycle re-armed short for the same reason.
2919
+ // Counting those toward the streak would be self-fulfilling —
2920
+ // each cycle that found the reason would park without attempting anything,
2921
+ // and after three the resident would sit parked-degraded for 6h at a
2922
+ // time. Never settled; streak reset.
2923
+ await this.ctx.storage.delete(DEGRADED_STREAK_KEY);
2924
+ }
2925
+ if (settled && (await this.isIdle())) {
2926
+ // isIdle awaited (git status per live tree) — re-read before writing.
2927
+ const now = (await this.ctx.storage.get<RepoFacts>(FACTS_KEY)) ?? facts;
2928
+ if (!now.idleSince)
2929
+ await this.ctx.storage.put(FACTS_KEY, {
2930
+ ...now,
2931
+ idleSince: new Date(systemClock()).toISOString(),
2932
+ } satisfies RepoFacts);
2933
+ this.rearmOutcome = "idle";
2934
+ return { go: false, why: "idle" }; // the alarm's finally re-arms at IDLE_REFRESH_INTERVAL_S
2935
+ }
2936
+ if (facts.idleSince) {
2937
+ const now = (await this.ctx.storage.get<RepoFacts>(FACTS_KEY)) ?? facts;
2938
+ const { idleSince: _woke, ...awake } = now;
2939
+ await this.ctx.storage.put(FACTS_KEY, awake satisfies RepoFacts);
2940
+ }
2941
+ return { go: true, record, facts };
2942
+ }
2943
+
2944
+ /** The cycle's fetch phase: mint, `refreshing`, fetch, the lockfile key at
2945
+ * the new tip. Token-mint failure is a command-level error — the resident
2946
+ * keeps serving the last snapshot and lifecycle state is NOT flipped by
2947
+ * it. It is recorded, and the cycle then CONTINUES with an anonymous
2948
+ * fetch (exactly what an unconfigured App does): a public repo outside
2949
+ * the installation stays fresh, and a private one fails at the fetch
2950
+ * below into a visible `degraded(github-unreachable: …)`. Returning early
2951
+ * instead would freeze whatever state the resident was in — a public
2952
+ * repo the App is not installed on would sit in the watchdog's
2953
+ * `degraded(alarm-missed)` forever with an ever-staler mirror, because
2954
+ * the App cannot mint for a repo it is not installed on. A failed fetch
2955
+ * is recorded here — `degraded` with its reason, the disk-full recovery
2956
+ * when that is the cause — and answered `ok: false`. */
2957
+ private async refreshFetch(
2958
+ resource: string,
2959
+ facts: RepoFacts,
2960
+ cycle: string,
2961
+ selfInFlight: number,
2962
+ ): Promise<
2963
+ | { ok: false; reason: string }
2964
+ | { ok: true; sha: string; lockfileHash: string; mintError: string | undefined; token: string | null }
2965
+ > {
2966
+ let token: string | null = null;
2967
+ // This cycle's mint error, kept so it survives the warm facts write
2968
+ // (which clears errors from PRIOR cycles) and prefixes a fetch failure's
2969
+ // reason — the observable for "App configured, repo outside the
2970
+ // installation" is a warm-but-anonymous resident with the mint named.
2971
+ let mintError: string | undefined;
2972
+ if (githubAppConfigured(this.env)) {
2973
+ try {
2974
+ token = (await mintRepoScopedToken(this.env, resource.slice("repo:".length))).token;
2975
+ } catch (err) {
2976
+ mintError = `token-mint-failed (command-level, fetching anonymously): ${errMsg(err)}`;
2977
+ await this.recordRefreshError(mintError);
2978
+ }
2979
+ }
2980
+
2981
+ await this.setResidentState("refreshing");
2982
+ let sha: string;
2983
+ try {
2984
+ // Same mirror mutex as attach's fetch/worktree work: the
2985
+ // refresh alarm and an in-flight attach serialize instead of racing
2986
+ // a prune against a worktree clone.
2987
+ sha = (await this.fetchMirror({ ref: facts.defaultRef, cycle, token })).sha;
2988
+ } catch (err) {
2989
+ // The fetch itself failed — or the tip could not be read afterwards,
2990
+ // which is the mirror's own failure, not GitHub's: the cycle's
2991
+ // classifier names that step.
2992
+ if (err instanceof StepError && err.step === "rev-parse") throw err;
2993
+ // A private repo whose mint failed lands here (the anonymous fetch is
2994
+ // refused): say so, rather than blaming GitHub reachability alone.
2995
+ const cause = mintError ? `${mintError}; then ` : "";
2996
+ const message = `${cause}${errMsg(err)}`;
2997
+ // A full disk fails this step too — the credential file is written
2998
+ // here (`ENOSPC` on /workspace/.resident/git-credentials would read as
2999
+ // github-unreachable, a SERVICEABLE reason, so every run would attach
3000
+ // and die at git-setup). Name the disk instead: not
3001
+ // serviceable, and the recovery below can free it.
3002
+ const failure = await this.classifyFailure("fetch", message);
3003
+ if (failure.diskFull) {
3004
+ await this.setResidentState("degraded", failure.reason);
3005
+ await this.recoverFromDiskFull(failure.reason, selfInFlight);
3006
+ return { ok: false, reason: failure.reason };
3007
+ }
3008
+ const reason = `github-unreachable: ${message}`;
3009
+ await this.setResidentState("degraded", reason);
3010
+ return { ok: false, reason };
3011
+ }
3012
+
3013
+ // Pure function of the commit — computed from the mirror before
3014
+ // any checkout work so the planner can compare it to the deps marker.
3015
+ const lockfileHash = sha === facts.sha ? facts.lockfileHash : await this.lockfileKey(sha);
3016
+ return { ok: true, sha, lockfileHash, mintError, token };
3017
+ }
3018
+
3019
+ /** A rebuild's first command: the markers for the steps about to be redone
3020
+ * come off — `built` always, `deps-key` only when the install runs — so
3021
+ * an interruption mid-step can never read as completion (item 48). */
3022
+ private async refreshClearMarkers(install: boolean): Promise<void> {
3023
+ await this.runOk(["rm", "-f", BUILT_MARKER, ...(install ? [DEPS_MARKER] : [])], "clear-markers");
3024
+ }
3025
+
3026
+ /** The cycle's install phase: the store entry for the new lockfile key,
3027
+ * bracketed by the installing marker (item 57) — written before, removed
3028
+ * by the build once the deps key lands, so a cycle that ends in between is
3029
+ * planned as a resume. The install is seeded from the key the checkout
3030
+ * holds now — npm reconciles the delta; a resumed install finds its key
3031
+ * already in the store when the last attempt completed, or reconciles
3032
+ * from the warm key again when it did not. Null when the command table
3033
+ * has no install: the build then links nothing. */
3034
+ private async refreshInstall(
3035
+ record: ResidentRecord,
3036
+ facts: RepoFacts,
3037
+ sha: string,
3038
+ lockfileHash: string,
3039
+ ): Promise<string | null> {
3040
+ if (!record.commands.install) return null;
3041
+ await this.writeDiskMarkers({ installingKey: lockfileHash });
3042
+ const { entry } = await this.installDeps({
3043
+ key: lockfileHash,
3044
+ sha,
3045
+ installCmd: record.commands.install,
3046
+ budgetMs: REFRESH_INSTALL_TIMEOUT_MS,
3047
+ seedFromKey: facts.lockfileHash,
3048
+ });
3049
+ return entry;
3050
+ }
3051
+
3052
+ /** The cycle's snapshot phase: the stamped pair to R2 and, when this cycle's
3053
+ * record stands (not superseded by another writer's), the ready marker and
3054
+ * the replaced snapshot's objects swept. Answers whether the record committed. */
3055
+ private async refreshSnapshot(resource: string, stamp: SnapshotStamp): Promise<boolean> {
3056
+ const snapped = await this.snapshot({ resource, stamp });
3057
+ const committed = snapped.done || !snapped.superseded;
3058
+ if (committed) {
3059
+ await this.writeDiskMarkers({ ready: stamp.sha });
3060
+ const previous = !snapped.done && !snapped.superseded ? snapped.previous : undefined;
3061
+ if (previous) await this.deleteBackupObjects([previous.mirror.id, previous.checkout.id]).catch(() => {});
3062
+ }
3063
+ return committed;
3064
+ }
3065
+
3066
+ /** The cycle's completion: the facts to the stamp (the snapshot step moved
3067
+ * them in the same write as the record — a wake never sees a half-updated
3068
+ * pair; this is the cycle's own bookkeeping on a fresh read, and a
3069
+ * superseded snapshot leaves the other writer's stamp alone), `warm`, then
3070
+ * the housekeeping that is never a lifecycle flip: the finished-ref
3071
+ * reclamation the prune already informed (item 45) and the disk sample
3072
+ * (item 55). */
3073
+ private async refreshComplete(
3074
+ resource: string,
3075
+ facts: RepoFacts,
3076
+ cycle: {
3077
+ sha: string;
3078
+ lockfileHash: string;
3079
+ committed: boolean;
3080
+ mintError: string | undefined;
3081
+ token: string | null;
3082
+ },
3083
+ ): Promise<void> {
3084
+ const fresh = (await this.ctx.storage.get<RepoFacts>(FACTS_KEY)) ?? facts;
3085
+ const updatedFacts: RepoFacts = {
3086
+ ...fresh,
3087
+ ...(cycle.committed
3088
+ ? { sha: cycle.sha, lockfileHash: cycle.lockfileHash, lastRefreshAt: new Date(systemClock()).toISOString() }
3089
+ : {}),
3090
+ };
3091
+ // Clear a PRIOR cycle's error; keep THIS cycle's mint error visible.
3092
+ delete updatedFacts.lastRefreshError;
3093
+ if (cycle.mintError) updatedFacts.lastRefreshError = cycle.mintError;
3094
+ // A wake cycle cleared idleSince at the gate; `facts` was read at entry and
3095
+ // still carries it — never resurrect it here (the dash would show a stale
3096
+ // "idle since" and every attach would take the wake-fetch path).
3097
+ delete updatedFacts.idleSince;
3098
+ await this.ctx.storage.put(FACTS_KEY, updatedFacts);
3099
+ await this.setResidentState("warm");
3100
+ // Event-triggered reclamation: the prune above already told the
3101
+ // mirror which branches died; finished refs give their worktree and
3102
+ // pool user back now, not at the idle TTL. Housekeeping, never a
3103
+ // lifecycle flip — a failure here is a log line.
3104
+ try {
3105
+ const gc = await this.reclaimFinishedRefs(resource, facts.defaultRef, cycle.token);
3106
+ if (gc.reclaimed.length > 0) console.log(`reclaim ${resource}: ${JSON.stringify(gc)}`);
3107
+ } catch (err) {
3108
+ console.log(`reclaim ${resource}: pass failed: ${errMsg(err)}`);
3109
+ }
3110
+ // Item 55: the cycle's disk sample — what /residents, `repo list`, the
3111
+ // watchdog line and the next attach admission read. Housekeeping too.
3112
+ await this.measureDisk().catch((err) => console.log(`disk: measure failed: ${errMsg(err)}`));
3113
+ }
3114
+
3115
+ /** What a cycle's throw means. A step killed from OUTSIDE (the container
3116
+ * replaced under it — an image-changing deploy or a container stop; a
3117
+ * Worker-only deploy leaves the container running and interrupts nothing)
3118
+ * is `refresh-interrupted`: not evidence about the repo — it never counts
3119
+ * toward the park streak (the entry gate) — and the chain re-arms SHORT so
3120
+ * the resident is warm again within a minute instead of after the full
3121
+ * cadence (an unclassified kill otherwise costs the resident the whole
3122
+ * 10-minute cadence, e.g. `degraded(build-failed: exit 143 …)` until the
3123
+ * next alarm); the instance throws it to the engine, whose retry re-enters
3124
+ * the step. Any other failure is the repo's own: `<step>-failed: …` /
3125
+ * `refresh-failed: …` as before. Non-StepErrors classify too — an SDK
3126
+ * replacement error can surface between steps — with the generic
3127
+ * "refresh" step, whose failure reason is the pre-existing
3128
+ * `refresh-failed: …` shape. A full disk is a third class: `disk-full: …`,
3129
+ * never serviceable, and the one failure the resident can act on itself
3130
+ * (recoverFromDiskFull). */
3131
+ private async classifyCycleError(err: unknown): Promise<RefreshFailure> {
3132
+ return err instanceof StepError
3133
+ ? await this.classifyFailure(err.step, err.message)
3134
+ : await this.classifyFailure("refresh", errMsg(err));
3135
+ }
3136
+
3137
+ /** Record a cycle's failure the one way: the classified reason in the log
3138
+ * (a StepError logged its own output block, but a failure between steps —
3139
+ * an SDK error, the markers, the snapshot — reached only the state entry,
3140
+ * which the next cycle's failure overwrites), on the facts (the degraded
3141
+ * state write can be clobbered within seconds by a concurrent attach/exec
3142
+ * whose ensureHydrated flips the state to `restoring · rehydrating`;
3143
+ * `lastRefreshError` survives that race and the next completed cycle
3144
+ * clears it, same as a mint error), then `degraded` with the last snapshot
3145
+ * still serving, and the disk-full recovery when that is the cause. */
3146
+ private async refreshFailed(failure: RefreshFailure, selfInFlight: number): Promise<void> {
3147
+ console.log(`refresh: cycle failed — ${failure.reason.slice(0, 400)}`);
3148
+ await this.recordRefreshError(failure.reason);
3149
+ await this.setResidentState("degraded", failure.reason); // last snapshot keeps serving
3150
+ if (failure.diskFull) await this.recoverFromDiskFull(failure.reason, selfInFlight);
3151
+ }
3152
+
3153
+ // -- the refresh cycle as a Workflow instance (item 7) --------------------------
3154
+ //
3155
+ // `ResidentRefresh` (the Workflow entrypoint, below the DO) calls these four
3156
+ // methods, one per step, through the DO stub. Each runs the same phase the
3157
+ // alarm runs, over the same rows, so a step the engine retries re-enters
3158
+ // the same idempotent read-then-act method (item 22) and finds the work
3159
+ // done. Inputs and answers are small facts — refs, shas, keys, a path, a
3160
+ // word — never a payload and never a credential: the token is minted inside
3161
+ // the step that needs it.
3162
+
3163
+ /** Which scheduler drives this resident's refresh cycle (LIFECYCLE_KEY). */
3164
+ async getLifecycle(): Promise<ResidentLifecycle> {
3165
+ return lifecycleOf(await this.ctx.storage.get(LIFECYCLE_KEY));
3166
+ }
3167
+
3168
+ /** Flip the resident between the alarm chain and the Workflow instance
3169
+ * (admin `/debug` `lifecycle`). To `workflow`: the pending alarm is
3170
+ * dropped; an alarm already firing runs to its end and re-arms nothing.
3171
+ * Back to `alarm`: the chain is re-armed the way the watchdog re-arms a
3172
+ * dead one, when the resident is in a state the chain serves. */
3173
+ async setLifecycle(mode: ResidentLifecycle): Promise<{ lifecycle: ResidentLifecycle; refreshSchedules: number }> {
3174
+ const previous = await this.getLifecycle();
3175
+ await this.ctx.storage.put(LIFECYCLE_KEY, mode);
3176
+ const resource = (await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "";
3177
+ if (mode === "workflow") {
3178
+ this.deleteSchedules(REFRESH_CALLBACK);
3179
+ } else if (previous !== "alarm") {
3180
+ const { state } = await this.getStatus();
3181
+ const pending = await this.listSchedules(REFRESH_CALLBACK);
3182
+ if (state !== "onboarding" && state !== "down" && pending.length === 0) {
3183
+ await this.schedule(5, REFRESH_CALLBACK, resource);
3184
+ }
3185
+ }
3186
+ console.log(`lifecycle: ${resource} ${previous} → ${mode}`);
3187
+ return { lifecycle: mode, refreshSchedules: (await this.listSchedules(REFRESH_CALLBACK)).length };
3188
+ }
3189
+
3190
+ private async instanceRow(): Promise<RefreshInstanceRow> {
3191
+ return (await this.ctx.storage.get<RefreshInstanceRow>(REFRESH_INSTANCE_KEY)) ?? { instance: null, skipped: null };
3192
+ }
3193
+
3194
+ /** The facts the cron's instance-creation decision reads
3195
+ * (`shouldCreateRefreshInstance`): the flag, the state and when it last
3196
+ * changed, idle mode, the last instance's creation time. */
3197
+ async refreshRow(): Promise<RefreshRow> {
3198
+ const map = await this.ctx.storage.get<unknown>([
3199
+ LIFECYCLE_KEY,
3200
+ STATE_KEY,
3201
+ UPDATED_KEY,
3202
+ FACTS_KEY,
3203
+ REFRESH_INSTANCE_KEY,
3204
+ ]);
3205
+ const facts = map.get(FACTS_KEY) as RepoFacts | undefined;
3206
+ const row = (map.get(REFRESH_INSTANCE_KEY) as RefreshInstanceRow | undefined) ?? { instance: null, skipped: null };
3207
+ const epochMs = (iso: string | undefined): number | null => {
3208
+ const t = Date.parse(iso ?? "");
3209
+ return Number.isFinite(t) ? t : null;
3210
+ };
3211
+ return {
3212
+ lifecycle: lifecycleOf(map.get(LIFECYCLE_KEY)),
3213
+ state: (map.get(STATE_KEY) as ResidentState | undefined) ?? "down",
3214
+ updatedAt: epochMs(map.get(UPDATED_KEY) as string | undefined),
3215
+ idleSince: epochMs(facts?.idleSince),
3216
+ lastInstanceAt: epochMs(row.instance?.createdAt),
3217
+ instanceRunning: await this.instanceRunning(row.instance?.id ?? null),
3218
+ };
3219
+ }
3220
+
3221
+ /** Whether the engine still runs `id`: queued, running, paused or waiting.
3222
+ * The one fact the marker's age cannot give — a step between retry
3223
+ * attempts holds no lease and writes nothing — read from the engine, which
3224
+ * knows. An unknown id, a missing binding or a failed read answer false:
3225
+ * the marker's age then decides, as before. */
3226
+ private async instanceRunning(id: string | null): Promise<boolean> {
3227
+ if (!id) return false;
3228
+ try {
3229
+ const { status } = await (await this.env.RESIDENT_REFRESH.get(id)).status();
3230
+ return (
3231
+ status === "queued" ||
3232
+ status === "running" ||
3233
+ status === "paused" ||
3234
+ status === "waiting" ||
3235
+ status === "waitingForPause"
3236
+ );
3237
+ } catch {
3238
+ return false;
3239
+ }
3240
+ }
3241
+
3242
+ /** The cron created an instance for this resident. */
3243
+ async recordRefreshInstance(id: string, createdAtMs: number): Promise<void> {
3244
+ const row = await this.instanceRow();
3245
+ await this.ctx.storage.put(REFRESH_INSTANCE_KEY, {
3246
+ ...row,
3247
+ instance: { id, createdAt: new Date(createdAtMs).toISOString(), lastStep: null, holder: null },
3248
+ } satisfies RefreshInstanceRow);
3249
+ }
3250
+
3251
+ /** The cron did not create for this bucket: a live cycle, or the engine's duplicate refusal. */
3252
+ async recordRefreshSkipped(id: string, atMs: number, why: string): Promise<void> {
3253
+ const row = await this.instanceRow();
3254
+ await this.ctx.storage.put(REFRESH_INSTANCE_KEY, {
3255
+ ...row,
3256
+ skipped: { id, at: new Date(atMs).toISOString(), why },
3257
+ } satisfies RefreshInstanceRow);
3258
+ }
3259
+
3260
+ /** The lifecycle flag and the instance row, for `/status` and `/debug info`. */
3261
+ async getRefreshView(): Promise<{
3262
+ lifecycle: ResidentLifecycle;
3263
+ instance: { id: string; createdAt: string; lastStep: string | null } | null;
3264
+ skipped: RefreshInstanceRow["skipped"];
3265
+ }> {
3266
+ const [lifecycle, row] = await Promise.all([this.getLifecycle(), this.instanceRow()]);
3267
+ const instance = row.instance
3268
+ ? { id: row.instance.id, createdAt: row.instance.createdAt, lastStep: row.instance.lastStep }
3269
+ : null;
3270
+ return { lifecycle, instance, skipped: row.skipped };
3271
+ }
3272
+
3273
+ /** An instance step ended: its outcome on the row, for the operator. An
3274
+ * instance the row does not know (created by hand) is adopted. */
3275
+ private async recordInstanceStep(instance: string, step: string, outcome: string): Promise<void> {
3276
+ const row = await this.instanceRow();
3277
+ const known = row.instance?.id === instance ? row.instance : null;
3278
+ await this.ctx.storage.put(REFRESH_INSTANCE_KEY, {
3279
+ ...row,
3280
+ instance: {
3281
+ id: instance,
3282
+ createdAt: known?.createdAt ?? new Date(systemClock()).toISOString(),
3283
+ holder: known?.holder ?? null,
3284
+ lastStep: residentText(`${step}: ${outcome}`),
3285
+ },
3286
+ } satisfies RefreshInstanceRow);
3287
+ }
3288
+
3289
+ /** The instance's fetch step took the cycle lease: remember the holder so a
3290
+ * later step — in this incarnation or the next — can release exactly it. */
3291
+ private async recordInstanceHolder(instance: string, holder: string): Promise<void> {
3292
+ const row = await this.instanceRow();
3293
+ const known = row.instance?.id === instance ? row.instance : null;
3294
+ await this.ctx.storage.put(REFRESH_INSTANCE_KEY, {
3295
+ ...row,
3296
+ instance: {
3297
+ id: instance,
3298
+ createdAt: known?.createdAt ?? new Date(systemClock()).toISOString(),
3299
+ lastStep: known?.lastStep ?? null,
3300
+ holder,
3301
+ },
3302
+ } satisfies RefreshInstanceRow);
3303
+ }
3304
+
3305
+ /** Release the cycle lease the instance holds, if any. */
3306
+ private async clearInstanceLease(instance: string): Promise<void> {
3307
+ const row = await this.instanceRow();
3308
+ if (row.instance?.id !== instance || !row.instance.holder) return;
3309
+ await this.clearInFlight("refresh", row.instance.holder);
3310
+ await this.ctx.storage.put(REFRESH_INSTANCE_KEY, {
3311
+ ...row,
3312
+ instance: { ...row.instance, holder: null },
3313
+ } satisfies RefreshInstanceRow);
3314
+ }
3315
+
3316
+ /** Run one step of the refresh instance the way the alarm runs its cycle:
3317
+ * counted in flight (so an attach-path reconcileImage never stops the
3318
+ * container under it), under a step trace the instance grafts on its root,
3319
+ * its outcome on the instance row for `/status`. A step killed from outside
3320
+ * (the container replaced under it) is thrown to the engine, whose retry
3321
+ * re-enters the same idempotent method — the row stays `refreshing`, never
3322
+ * `degraded`, and a `refreshing` younger than the stale bound keeps the
3323
+ * cron from creating a second instance meanwhile. A failure of the repo's
3324
+ * own is recorded as the alarm records it — `degraded` with the reason,
3325
+ * the last snapshot still serving — and answered `failed`, which ends the
3326
+ * instance; the next cron firing starts the next cycle from that state. */
3327
+ private async runInstanceStep<T>(
3328
+ instance: string,
3329
+ step: string,
3330
+ fn: () => Promise<InstanceStepResult<T>>,
3331
+ ): Promise<InstanceStepAnswer<T>> {
3332
+ const startedAt = systemClock();
3333
+ const trace = createStepTrace(startedAt);
3334
+ this.refreshesInFlight++;
3335
+ let outcome = "done";
3336
+ try {
3337
+ const result = await this.stepTrace.run(trace, fn);
3338
+ if (result.status !== "done") {
3339
+ outcome = result.status === "stopped" ? `stopped (${result.why})` : `failed (${result.reason})`;
3340
+ await this.clearInstanceLease(instance);
3341
+ }
3342
+ return { ...result, startedAt, trace: trace.steps() };
3343
+ } catch (err) {
3344
+ if (err instanceof ResidentDownError) {
3345
+ // Already down with its reason (goDown recorded it); the instance ends.
3346
+ outcome = `failed (${err.message})`;
3347
+ await this.clearInstanceLease(instance);
3348
+ return { status: "failed", reason: err.message, startedAt, trace: trace.steps() };
3349
+ }
3350
+ const failure = await this.classifyCycleError(err);
3351
+ if (failure.interrupted) {
3352
+ outcome = `interrupted (${failure.reason}) — the engine retries`;
3353
+ console.log(`refresh instance ${instance}: ${step} interrupted — ${failure.reason.slice(0, 400)}; retrying`);
3354
+ throw err;
3355
+ }
3356
+ await this.refreshFailed(failure, 1);
3357
+ await this.clearInstanceLease(instance);
3358
+ outcome = `failed (${failure.reason})`;
3359
+ return { status: "failed", reason: failure.reason, startedAt, trace: trace.steps() };
3360
+ } finally {
3361
+ this.refreshesInFlight--;
3362
+ // The gates set this for the alarm's finally; no alarm runs on this path.
3363
+ this.rearmOutcome = "normal";
3364
+ await this.recordInstanceStep(instance, step, outcome).catch((err) =>
3365
+ console.log(`refresh instance ${instance}: recording ${step} failed: ${errMsg(err)}`),
3366
+ );
2575
3367
  }
2576
3368
  }
2577
3369
 
3370
+ /** Step `fetch`: the gates, the cycle lease, the fetch and the plan. The
3371
+ * instance id is the cycle `fetchMirror` records, so a retry of this step
3372
+ * finds its fetch done. A rebuild's markers come off here, once per cycle,
3373
+ * never at the build step — a retried build must find its own work done. */
3374
+ async refreshInstanceFetch(input: {
3375
+ resource: string;
3376
+ instance: string;
3377
+ }): Promise<InstanceStepAnswer<RefreshFetchFacts>> {
3378
+ return this.runInstanceStep<RefreshFetchFacts>(input.instance, "fetch", async () => {
3379
+ const before = await this.getStatus();
3380
+ // down stays down (a rebuild is the escape hatch); onboarding is owned by provisioning.
3381
+ if (before.state === "onboarding" || before.state === "down") return { status: "stopped", why: "not-serving" };
3382
+ const gate = await this.refreshGate(input.resource);
3383
+ if (!gate.go) return { status: "stopped", why: gate.why };
3384
+ const { record, facts } = gate;
3385
+ const holder = this.nextHolder();
3386
+ await this.recordInFlight("refresh", holder, REFRESH_CYCLE_LEASE_MS, "refresh");
3387
+ await this.recordInstanceHolder(input.instance, holder);
3388
+ const fetched = await this.refreshFetch(input.resource, facts, input.instance, 1);
3389
+ if (!fetched.ok) return { status: "failed", reason: fetched.reason };
3390
+ const { sha, lockfileHash } = fetched;
3391
+ const plan = planRefresh({
3392
+ sha,
3393
+ factsSha: facts.sha,
3394
+ lockfileKey: lockfileHash,
3395
+ disk: await this.readRefreshDisk(),
3396
+ });
3397
+ if (plan.action !== "unchanged") {
3398
+ console.log(`refresh: ${facts.sha.slice(0, 8)} → ${sha.slice(0, 8)}: ${plan.action} (${plan.why})`);
3399
+ }
3400
+ if (plan.action === "rebuild") await this.refreshClearMarkers(plan.install);
3401
+ return {
3402
+ status: "done",
3403
+ ref: facts.defaultRef,
3404
+ sha,
3405
+ factsSha: facts.sha,
3406
+ lockfileKey: lockfileHash,
3407
+ action: plan.action,
3408
+ install: plan.action === "rebuild" && plan.install && !!record.commands.install,
3409
+ mintError: fetched.mintError ?? null,
3410
+ };
3411
+ });
3412
+ }
3413
+
3414
+ /** Step `install`: the store entry for the new key (`installDeps` finds a
3415
+ * complete entry done). Answers the entry's path for the build to link. */
3416
+ async refreshInstanceInstall(input: {
3417
+ resource: string;
3418
+ instance: string;
3419
+ sha: string;
3420
+ lockfileKey: string;
3421
+ }): Promise<InstanceStepAnswer<{ entry: string | null }>> {
3422
+ return this.runInstanceStep<{ entry: string | null }>(input.instance, "install", async () => {
3423
+ const record = await this.registry().getRecord(input.resource);
3424
+ if (!record) return { status: "stopped", why: "offboarded" };
3425
+ const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
3426
+ if (!facts) return { status: "stopped", why: "no-facts" };
3427
+ const entry = await this.refreshInstall(record, facts, input.sha, input.lockfileKey);
3428
+ return { status: "done", entry };
3429
+ });
3430
+ }
3431
+
3432
+ /** Step `build`: the checkout to the sha and its build (`runBuild` finds a
3433
+ * checkout whose markers all name the target done). */
3434
+ async refreshInstanceBuild(input: {
3435
+ resource: string;
3436
+ instance: string;
3437
+ sha: string;
3438
+ factsSha: string;
3439
+ lockfileKey: string;
3440
+ depsEntry: string | null;
3441
+ }): Promise<InstanceStepAnswer<{ ran: boolean; why: string }>> {
3442
+ return this.runInstanceStep<{ ran: boolean; why: string }>(input.instance, "build", async () => {
3443
+ const record = await this.registry().getRecord(input.resource);
3444
+ if (!record) return { status: "stopped", why: "offboarded" };
3445
+ const built = await this.runBuild({
3446
+ sha: input.sha,
3447
+ factsSha: input.factsSha,
3448
+ lockfileKey: input.lockfileKey,
3449
+ buildCmd: record.commands.build,
3450
+ depsEntry: input.depsEntry,
3451
+ });
3452
+ // `done` from the plan means the tree was already built for this sha; the
3453
+ // step reports whether a build actually ran.
3454
+ return { status: "done", ran: !built.done, why: built.why };
3455
+ });
3456
+ }
3457
+
3458
+ /** Step `snapshot`: the stamped pair to R2 (`snapshot` finds a record at the
3459
+ * stamp done and answers `superseded` to another writer, never a throw),
3460
+ * then the cycle's completion — facts, `warm`, the reclamation, the disk
3461
+ * sample — and the cycle lease released. An `unchanged` cycle skips the
3462
+ * archive and still completes, as the alarm does. */
3463
+ async refreshInstanceSnapshot(input: {
3464
+ resource: string;
3465
+ instance: string;
3466
+ ref: string;
3467
+ sha: string;
3468
+ lockfileKey: string;
3469
+ action: RefreshPlan["action"];
3470
+ mintError: string | null;
3471
+ }): Promise<InstanceStepAnswer<{ committed: boolean }>> {
3472
+ return this.runInstanceStep<{ committed: boolean }>(input.instance, "snapshot", async () => {
3473
+ const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
3474
+ if (!facts) return { status: "stopped", why: "no-facts" };
3475
+ const committed =
3476
+ input.action === "unchanged"
3477
+ ? true
3478
+ : await this.refreshSnapshot(input.resource, {
3479
+ ref: input.ref,
3480
+ sha: input.sha,
3481
+ lockfileHash: input.lockfileKey,
3482
+ });
3483
+ // The reclamation's token: minted here (cached per slug), never carried
3484
+ // between steps. A failed mint runs the reclamation anonymously (PR lookups
3485
+ // answer unknown, trees are kept) and is recorded like the fetch's.
3486
+ let token: string | null = null;
3487
+ let snapshotMintError: string | undefined;
3488
+ if (githubAppConfigured(this.env)) {
3489
+ try {
3490
+ token = (await mintRepoScopedToken(this.env, input.resource.slice("repo:".length))).token;
3491
+ } catch (err) {
3492
+ snapshotMintError = `token-mint-failed (reclamation runs anonymously): ${errMsg(err)}`;
3493
+ console.log(`refresh instance ${input.instance}: ${snapshotMintError}`);
3494
+ }
3495
+ }
3496
+ await this.refreshComplete(input.resource, facts, {
3497
+ sha: input.sha,
3498
+ lockfileHash: input.lockfileKey,
3499
+ committed,
3500
+ mintError: input.mintError ?? snapshotMintError,
3501
+ token,
3502
+ });
3503
+ await this.clearInstanceLease(input.instance);
3504
+ return { status: "done", committed };
3505
+ });
3506
+ }
3507
+
2578
3508
  /** Set during one alarm by the idle gate (`idle`), an image-stale container
2579
3509
  * stop (`image-stale-restart`), a disk-full recycle (`disk-full-restart`)
2580
3510
  * or an interrupted step (`interrupted`) so `finally` picks the matching
@@ -2658,7 +3588,7 @@ export class ResidentDO extends Sandbox<Env> {
2658
3588
  );
2659
3589
  await this.ctx.storage.put(DISK_FULL_RECYCLE_KEY, systemClock());
2660
3590
  await this.recordRefreshError(`${reason} — container recycled; restoring from R2 on the next alarm`);
2661
- this.clearIncarnationMemos(); // deliberate incarnation swap
3591
+ this.swapIncarnation(); // deliberate incarnation swap
2662
3592
  await this.stop().catch((err) => console.log(`disk-full: stop failed: ${errMsg(err)}`));
2663
3593
  this.rearmOutcome = "disk-full-restart";
2664
3594
  }
@@ -2941,7 +3871,7 @@ export class ResidentDO extends Sandbox<Env> {
2941
3871
  console.log(
2942
3872
  `image-stale (${where}): ${last} missing in the running container — stopping so it restarts on the current image`,
2943
3873
  );
2944
- this.clearIncarnationMemos(); // deliberate incarnation swap
3874
+ this.swapIncarnation(); // deliberate incarnation swap
2945
3875
  await this.stop().catch((err) => console.log(`image-stale: stop failed: ${errMsg(err)}`));
2946
3876
  return true;
2947
3877
  }
@@ -2962,9 +3892,11 @@ export class ResidentDO extends Sandbox<Env> {
2962
3892
  action: "none" | "rearmed" | "provision-timed-out" | "auto-rebuilt";
2963
3893
  /** Item 55: the last disk sample's gauge, for the watchdog's status line. */
2964
3894
  disk: { usedKiB: number; totalKiB: number; freeKiB: number; at: string } | null;
3895
+ /** Item 7: what the cron's instance-creation decision reads, after the check above settled the state. */
3896
+ refresh: RefreshRow;
2965
3897
  }> {
2966
3898
  const [check, disk] = await Promise.all([this.watchdogCheckLifecycle(), this.diskGauge()]);
2967
- return { ...check, disk };
3899
+ return { ...check, disk, refresh: await this.refreshRow() };
2968
3900
  }
2969
3901
 
2970
3902
  private async watchdogCheckLifecycle(): Promise<{
@@ -3007,6 +3939,10 @@ export class ResidentDO extends Sandbox<Env> {
3007
3939
  // Any serving state clears accumulated strikes (a recovery must reset the
3008
3940
  // counter, or an unrelated later down inherits stale strikes).
3009
3941
  await this.ctx.storage.delete(REBUILD_STRIKES_KEY);
3942
+ // Item 7: a resident on the Workflow lifecycle has no chain to re-arm —
3943
+ // its cycles are the instances the cron creates. The sweep re-arm below
3944
+ // is housekeeping either way; the two refresh re-arms are the alarm's.
3945
+ const chained = (await this.getLifecycle()) === "alarm";
3010
3946
 
3011
3947
  // The sweep chain has the same failure mode as the refresh chain (a DO
3012
3948
  // eviction mid-callback kills the self-rescheduling), but nothing re-armed
@@ -3050,28 +3986,49 @@ export class ResidentDO extends Sandbox<Env> {
3050
3986
  // named degradation, never a stall) and pull the next cycle to +5s so it normalizes.
3051
3987
  if (status.state === "refreshing" || status.state === "restoring") {
3052
3988
  const updatedAt = Date.parse((await this.ctx.storage.get<string>(UPDATED_KEY)) ?? "") || 0;
3053
- // A hydration older than the stale bound counts as DEAD, not in flight:
3054
- // its promise lives on SDK calls into a container that may have
3055
- // been replaced under it, and a promise that never settles would
3056
- // otherwise hold `this.hydration` non-null forever — making a stuck
3057
- // `restoring` permanently invisible to this branch. No legitimate
3989
+ // Who holds what comes from the in-flight row (item 22), not from this
3990
+ // isolate's memory: a cycle or hydration lease is alive only for the
3991
+ // current incarnation and inside its budget. A hydration past the stale
3992
+ // bound counts as DEAD, not in flight: its promise lives on SDK calls
3993
+ // into a container that may have been replaced under it, and a promise
3994
+ // that never settles would otherwise hold the memo forever — making a
3995
+ // stuck `restoring` permanently invisible to this branch. No legitimate
3058
3996
  // restore approaches STALE_MIDFLIGHT_MS (a full R2 restore is ~1 min).
3059
- const hydrationLive = this.hydration !== null && systemClock() - this.hydrationStartedAt <= STALE_MIDFLIGHT_MS;
3060
- const inFlight = this.refreshesInFlight > 0 || hydrationLive;
3997
+ const live = liveInFlight(await this.readInFlight(), systemClock(), this.incarnation);
3998
+ const inFlight = live.refresh || live.hydration;
3061
3999
  if (!inFlight && systemClock() - updatedAt > STALE_MIDFLIGHT_MS) {
3062
4000
  // The reads above yielded; a cycle that started meanwhile owns the
3063
4001
  // state now — leave it alone rather than stamp `degraded` over it.
3064
4002
  const again = await this.getStatus();
3065
- const hydrationStillDead =
3066
- this.hydration === null || systemClock() - this.hydrationStartedAt > STALE_MIDFLIGHT_MS;
3067
- if (again.state !== status.state || this.refreshesInFlight > 0 || !hydrationStillDead) {
4003
+ const rowAgain = await this.readInFlight();
4004
+ const liveAgain = liveInFlight(rowAgain, systemClock(), this.incarnation);
4005
+ if (again.state !== status.state || liveAgain.refresh || liveAgain.hydration) {
3068
4006
  return { resource, ...again, action: "none" };
3069
4007
  }
3070
- // Drop the dead hydration reference so the re-armed cycle's
3071
- // ensureHydrated starts a fresh restore instead of awaiting a promise
3072
- // that will never settle. Safe: past the bound nothing on the other
3073
- // end is still writing (the container it talked to is gone).
4008
+ // Drop the dead hydration reference and the dead leases so the
4009
+ // re-armed cycle's ensureHydrated starts a fresh restore instead of
4010
+ // awaiting a promise that will never settle. Safe: past the bound
4011
+ // nothing on the other end is still writing (the container it talked
4012
+ // to is gone). Each clear is compared against the holder just read:
4013
+ // a cycle that recorded a fresh lease between that read and this
4014
+ // delete keeps it, the way a release never deletes another holder's row.
3074
4015
  this.hydration = null;
4016
+ if (!chained && (await this.instanceRunning((await this.instanceRow()).instance?.id ?? null))) {
4017
+ // The engine still runs the recorded instance — a step between retry
4018
+ // attempts, holding no lease and writing no state. Not stale: leave
4019
+ // the marker, create nothing (the cron's decision reads the same fact).
4020
+ return { resource, ...status, action: "none" };
4021
+ }
4022
+ if (rowAgain.refresh) await this.clearInFlight("refresh", rowAgain.refresh.holder);
4023
+ if (rowAgain.hydration) await this.clearInFlight("hydration", rowAgain.hydration.holder);
4024
+ if (!chained) {
4025
+ // The orphan is named the same way; the next instance the cron
4026
+ // creates (this very pass — the marker is no longer `refreshing`)
4027
+ // normalizes it, no alarm involved.
4028
+ const reason = `stale-mid-flight: ${status.state} since ${new Date(updatedAt).toISOString()} with no cycle running; the next refresh instance normalizes it`;
4029
+ await this.setResidentState("degraded", reason);
4030
+ return { resource, state: "degraded", reason, action: "none" };
4031
+ }
3075
4032
  const reason = `stale-mid-flight: ${status.state} since ${new Date(updatedAt).toISOString()} with no cycle running; re-armed by watchdog`;
3076
4033
  await this.setResidentState("degraded", reason);
3077
4034
  this.deleteSchedules(REFRESH_CALLBACK);
@@ -3080,6 +4037,8 @@ export class ResidentDO extends Sandbox<Env> {
3080
4037
  }
3081
4038
  }
3082
4039
 
4040
+ // No chain to be dead under the Workflow lifecycle (item 7).
4041
+ if (!chained) return { resource, ...status, action: "none" };
3083
4042
  const pending = await this.listSchedules(REFRESH_CALLBACK);
3084
4043
  if (pending.length === 0) {
3085
4044
  await this.schedule(5, REFRESH_CALLBACK, resource);
@@ -3788,8 +4747,9 @@ export class ResidentDO extends Sandbox<Env> {
3788
4747
  * `restoring` span stays under RESTORE_MAX_MS in total (the hydrate
3789
4748
  * invariant); every other caller gets `min(budgetMs, RESTORE_MAX_MS)`
3790
4749
  * from now, so an attach never waits longer for a download than it
3791
- * would for an install. */
3792
- opts: { seedFromKey?: string; restoreDeadlineMs?: number } = {},
4750
+ * would for an install. `attempt`: the private scratch tree's name when
4751
+ * the caller has leased it (installDeps); minted here otherwise. */
4752
+ opts: { seedFromKey?: string; restoreDeadlineMs?: number; attempt?: string } = {},
3793
4753
  ): Promise<string> {
3794
4754
  const backupRecord = await this.depsBackupRecord(key);
3795
4755
  const plan = planDepsMaterialization({
@@ -3808,14 +4768,15 @@ export class ResidentDO extends Sandbox<Env> {
3808
4768
  // same way next time) and falls through to the installer, which records
3809
4769
  // a fresh backup — never a stranded key.
3810
4770
  const restoreDeadlineMs = opts.restoreDeadlineMs ?? systemClock() + Math.min(budgetMs, RESTORE_MAX_MS);
4771
+ const attempt = opts.attempt ?? crypto.randomUUID().slice(0, 8);
3811
4772
  const p = (
3812
4773
  plan.action === "restore" && backupRecord
3813
- ? this.restoreDepsEntry(key, backupRecord, restoreDeadlineMs).catch(async (err) => {
4774
+ ? this.restoreDepsEntry(key, backupRecord, restoreDeadlineMs, attempt).catch(async (err) => {
3814
4775
  console.log(`deps: restore of ${key.slice(0, 8)} failed — installing instead: ${errMsg(err)}`);
3815
4776
  await this.dropDepsBackups([key]).catch(() => {});
3816
- return this.installDepsEntry(key, sha, installCmd, budgetMs, opts);
4777
+ return this.installDepsEntry(key, sha, installCmd, budgetMs, { ...opts, attempt });
3817
4778
  })
3818
- : this.installDepsEntry(key, sha, installCmd, budgetMs, opts)
4779
+ : this.installDepsEntry(key, sha, installCmd, budgetMs, { ...opts, attempt })
3819
4780
  ).finally(() => this.depsInFlight.delete(key));
3820
4781
  this.depsInFlight.set(key, p);
3821
4782
  return p;
@@ -3875,8 +4836,12 @@ export class ResidentDO extends Sandbox<Env> {
3875
4836
  * then the same commit script an install ends with: staging, atomic
3876
4837
  * rename, `.complete` LAST. A partial download never becomes an entry.
3877
4838
  * `deadlineMs` is the caller's (see materializeDeps): never a fresh cap. */
3878
- private async restoreDepsEntry(key: string, record: DepsBackupRecord, deadlineMs: number): Promise<string> {
3879
- const attempt = crypto.randomUUID().slice(0, 8);
4839
+ private async restoreDepsEntry(
4840
+ key: string,
4841
+ record: DepsBackupRecord,
4842
+ deadlineMs: number,
4843
+ attempt: string,
4844
+ ): Promise<string> {
3880
4845
  const scratch = depsScratchPath(attempt);
3881
4846
  const staging = depsStagingPath(key, attempt);
3882
4847
  const t0 = systemClock();
@@ -3933,10 +4898,10 @@ export class ResidentDO extends Sandbox<Env> {
3933
4898
  sha: string,
3934
4899
  installCmd: string,
3935
4900
  budgetMs: number,
3936
- opts: { seedFromKey?: string },
4901
+ opts: { seedFromKey?: string; attempt: string },
3937
4902
  ): Promise<string> {
3938
4903
  await this.acquireDepsInstallSlot();
3939
- const attempt = crypto.randomUUID().slice(0, 8);
4904
+ const { attempt } = opts;
3940
4905
  const scratch = depsScratchPath(attempt);
3941
4906
  const staging = depsStagingPath(key, attempt);
3942
4907
  const t0 = systemClock();
@@ -5145,10 +6110,19 @@ export class ResidentDO extends Sandbox<Env> {
5145
6110
  FACTS_KEY,
5146
6111
  SNAPSHOT_KEY,
5147
6112
  DISK_KEY,
6113
+ MIRROR_MUTEX_KEY,
6114
+ inFlightKey("refresh"),
6115
+ inFlightKey("hydration"),
6116
+ LIFECYCLE_KEY,
6117
+ REFRESH_INSTANCE_KEY,
5148
6118
  ]);
5149
6119
  const facts = map.get(FACTS_KEY) as RepoFacts | undefined;
5150
6120
  const snap = map.get(SNAPSHOT_KEY) as SnapshotRecord | undefined;
5151
6121
  const disk = (map.get(DISK_KEY) as DiskSample | undefined) ?? null;
6122
+ const refreshRow = (map.get(REFRESH_INSTANCE_KEY) as RefreshInstanceRow | undefined) ?? {
6123
+ instance: null,
6124
+ skipped: null,
6125
+ };
5152
6126
  const [refresh, provisionRun, provisionDeadline, bindings] = await Promise.all([
5153
6127
  this.listSchedules(REFRESH_CALLBACK),
5154
6128
  this.listSchedules(PROVISION_RUN_CALLBACK),
@@ -5203,6 +6177,28 @@ export class ResidentDO extends Sandbox<Env> {
5203
6177
  inFlight: this.inFlightCount(),
5204
6178
  // The runs alone (no refresh cycle): what the deploy preflight refuses on.
5205
6179
  runsInFlight: this.runsInFlightCount(),
6180
+ // Item 22: who holds what, as the rows say — the mirror mutex and the
6181
+ // cycle/hydration leases, each judged against this incarnation.
6182
+ incarnation: this.incarnation,
6183
+ mirrorMutex: (map.get(MIRROR_MUTEX_KEY) as Lease | undefined) ?? null,
6184
+ leases: inFlightRow(
6185
+ map.get(inFlightKey("refresh")) as Lease | undefined,
6186
+ map.get(inFlightKey("hydration")) as Lease | undefined,
6187
+ ),
6188
+ // Item 7: which scheduler drives the refresh cycle, and — on the
6189
+ // Workflow lifecycle — the instance the cron last created with the step
6190
+ // it last reported, and the last bucket the cron skipped.
6191
+ lifecycle: lifecycleOf(map.get(LIFECYCLE_KEY)),
6192
+ refresh: {
6193
+ instance: refreshRow.instance
6194
+ ? {
6195
+ id: refreshRow.instance.id,
6196
+ createdAt: refreshRow.instance.createdAt,
6197
+ lastStep: refreshRow.instance.lastStep,
6198
+ }
6199
+ : null,
6200
+ skipped: refreshRow.skipped,
6201
+ },
5206
6202
  threads,
5207
6203
  // Item 55: the last disk sample (`residentDiskBudget.ts` DiskSample), or
5208
6204
  // null before the first measurement of this incarnation.
@@ -5228,12 +6224,16 @@ export class ResidentDO extends Sandbox<Env> {
5228
6224
  return { killed: true, remaining: (await this.listSchedules(REFRESH_CALLBACK)).length };
5229
6225
  }
5230
6226
 
5231
- /** Pull the next refresh forward to ~1s from now. */
5232
- async debugRefreshNow(): Promise<{ scheduled: boolean }> {
6227
+ /** Pull the next refresh forward to ~1s from now. A resident on the Workflow
6228
+ * lifecycle has no chain to pull: its next cycle is the instance the next
6229
+ * cron firing creates (`run-watchdog` runs that pass on demand). */
6230
+ async debugRefreshNow(): Promise<{ scheduled: boolean; lifecycle: ResidentLifecycle }> {
6231
+ const lifecycle = await this.getLifecycle();
6232
+ if (lifecycle === "workflow") return { scheduled: false, lifecycle };
5233
6233
  const resource = (await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "";
5234
6234
  this.deleteSchedules(REFRESH_CALLBACK);
5235
6235
  await this.schedule(1, REFRESH_CALLBACK, resource);
5236
- return { scheduled: true };
6236
+ return { scheduled: true, lifecycle };
5237
6237
  }
5238
6238
 
5239
6239
  /** Fault injection for the watchdog's stuck-onboarding path: re-persist
@@ -5250,7 +6250,7 @@ export class ResidentDO extends Sandbox<Env> {
5250
6250
  * next refresh must take the restoring→warm wake path). */
5251
6251
  async debugStopContainer(): Promise<{ stopped: boolean; error?: string }> {
5252
6252
  try {
5253
- this.clearIncarnationMemos(); // deliberate incarnation swap
6253
+ this.swapIncarnation(); // deliberate incarnation swap
5254
6254
  await this.stop();
5255
6255
  return { stopped: true };
5256
6256
  } catch (err) {
@@ -5423,7 +6423,7 @@ export class ResidentDO extends Sandbox<Env> {
5423
6423
  }
5424
6424
  // Retired DO: clear the alarm the Container base may have armed for its
5425
6425
  // schedules, then wipe storage so nothing ever wakes this object again.
5426
- this.clearIncarnationMemos(); // retired object, retired memos
6426
+ this.swapIncarnation(); // retired object, retired memos
5427
6427
  await this.ctx.storage.deleteAlarm();
5428
6428
  await this.ctx.storage.deleteAll();
5429
6429
  return { schedulesCancelled: true, containerStopped, storageCleared: true, backupObjectsDeleted, errors };
@@ -6213,13 +7213,22 @@ async function handleStatus(env: Env, url: URL): Promise<Response> {
6213
7213
  // deploy gate reads /residents. The registry check rides in the same flight
6214
7214
  // (its 404 is judged first, the probes' results discarded then).
6215
7215
  const stub = residentStub(env, resource.resource);
6216
- const [record, status, inFlight] = await Promise.all([
7216
+ const [record, status, inFlight, refresh] = await Promise.all([
6217
7217
  registryStub(env).getRecord(resource.resource),
6218
7218
  stub.getStatus(),
6219
7219
  stub.getInFlightCount(),
7220
+ stub.getRefreshView(),
6220
7221
  ]);
6221
7222
  if (!record) return json({ error: `${resource.resource} is not onboarded` }, 404);
6222
- return json({ state: status.state, reason: status.reason, inFlight });
7223
+ // Item 7: which scheduler drives the refresh cycle and, on the Workflow
7224
+ // lifecycle, the current instance with its last step and the last skipped bucket.
7225
+ return json({
7226
+ state: status.state,
7227
+ reason: status.reason,
7228
+ inFlight,
7229
+ lifecycle: refresh.lifecycle,
7230
+ refresh: { instance: refresh.instance, skipped: refresh.skipped },
7231
+ });
6223
7232
  }
6224
7233
 
6225
7234
  // -- thread data plane handlers -----------------------------------------------
@@ -6578,20 +7587,87 @@ async function handleDebug(env: Env, body: Record<string, unknown>): Promise<Res
6578
7587
  if ("error" in days) return json({ error: days.error }, 400);
6579
7588
  return json(await stub.debugBackdateThread(threadKey.threadKey, days.value));
6580
7589
  }
7590
+ case "lifecycle": {
7591
+ // Item 7: which scheduler drives this resident's refresh cycle. Admin
7592
+ // scope, one resident at a time; the default for every resident is `alarm`.
7593
+ const mode = parseLifecycle(body.mode);
7594
+ if (!mode) return json({ error: 'mode must be "alarm" or "workflow"' }, 400);
7595
+ return json({ op, resource: resource.resource, ...(await stub.setLifecycle(mode)) });
7596
+ }
6581
7597
  default:
6582
7598
  return json(
6583
7599
  {
6584
- error: `unknown op ${JSON.stringify(op)} (ops: info, schedules, kill-refresh, refresh-now, stop-container, force-onboarding, force-down, mint-token, run-watchdog, set-test-overrides, threads, sweep-now, reclaim-now, measure-disk, purge-bindings, backdate-thread)`,
7600
+ error: `unknown op ${JSON.stringify(op)} (ops: info, schedules, kill-refresh, refresh-now, stop-container, force-onboarding, force-down, mint-token, run-watchdog, set-test-overrides, threads, sweep-now, reclaim-now, measure-disk, purge-bindings, backdate-thread, lifecycle)`,
6585
7601
  },
6586
7602
  400,
6587
7603
  );
6588
7604
  }
6589
7605
  }
6590
7606
 
7607
+ /** The cron's instance-creation duty for one resident (item 7). Only a
7608
+ * `workflow` row gets an instance, at most one per ten-minute bucket, never
7609
+ * while a cycle is live (`shouldCreateRefreshInstance`); the id is
7610
+ * deterministic per resident and bucket, so a second firing in one bucket
7611
+ * meets the engine's duplicate-id refusal, which is the expected no-op. A
7612
+ * skipped live cycle and a duplicate are recorded on the row for `/status`.
7613
+ * An `alarm` resident answers null: nothing here touches it. */
7614
+ /** Whether the engine knows an instance by this id, in any status. A missing
7615
+ * id rejects on `get` or on `status`; either way the answer is false. */
7616
+ async function refreshInstanceExists(env: Env, id: string): Promise<boolean> {
7617
+ try {
7618
+ await (await env.RESIDENT_REFRESH.get(id)).status();
7619
+ return true;
7620
+ } catch {
7621
+ return false;
7622
+ }
7623
+ }
7624
+
7625
+ async function createRefreshInstance(
7626
+ env: Env,
7627
+ stub: ReturnType<typeof residentStub>,
7628
+ resource: string,
7629
+ row: RefreshRow,
7630
+ ): Promise<RefreshInstanceAction | null> {
7631
+ if (row.lifecycle !== "workflow") return null;
7632
+ const now = systemClock();
7633
+ const decision = shouldCreateRefreshInstance(row, now, {
7634
+ intervalS: REFRESH_INTERVAL_S,
7635
+ idleIntervalS: IDLE_REFRESH_INTERVAL_S,
7636
+ });
7637
+ const slug = resource.slice("repo:".length);
7638
+ const slash = slug.indexOf("/");
7639
+ const id = refreshInstanceId(slug.slice(0, slash), slug.slice(slash + 1), now);
7640
+ if (!decision.create) {
7641
+ if (decision.why === "mid-cycle" || decision.why === "running")
7642
+ await stub.recordRefreshSkipped(id, now, decision.why);
7643
+ return { id, action: "skipped", why: decision.why };
7644
+ }
7645
+ try {
7646
+ await env.RESIDENT_REFRESH.create({ id, params: { resource } });
7647
+ } catch (err) {
7648
+ const message = errMsg(err);
7649
+ // The engine refuses an id that names an instance still inside its
7650
+ // retention, and the refusal carries no code — so the id is asked, not
7651
+ // the wording: an instance that answers for it exists, and the refusal
7652
+ // was the duplicate it looks like. Any other failure stays a failure.
7653
+ if (await refreshInstanceExists(env, id)) {
7654
+ await stub.recordRefreshSkipped(id, now, "duplicate");
7655
+ return { id, action: "duplicate", why: "duplicate" };
7656
+ }
7657
+ console.error(`resident-watchdog: creating refresh instance ${id} failed — ${message}`);
7658
+ return { id, action: "failed", why: residentText(message) };
7659
+ }
7660
+ await stub.recordRefreshInstance(id, now);
7661
+ console.log(`resident-watchdog: created refresh instance ${id}`);
7662
+ return { id, action: "created", why: decision.why };
7663
+ }
7664
+
6591
7665
  /** One watchdog pass over every registered resident. Shared by the cron
6592
7666
  * handler and the /debug run-watchdog op. Each check targets a different DO,
6593
7667
  * so they run concurrently; a failing one becomes its own {error} entry
6594
- * without touching its neighbors, and the results follow the registry list. */
7668
+ * without touching its neighbors, and the results follow the registry list.
7669
+ * For a resident on the Workflow lifecycle the pass also creates the refresh
7670
+ * instance the bucket is due (item 7). */
6595
7671
  async function runWatchdog(env: Env, parent?: TraceSpan): Promise<WatchdogSummary> {
6596
7672
  const registry = registryStub(env);
6597
7673
  const residents = await registry.list();
@@ -6599,13 +7675,15 @@ async function runWatchdog(env: Env, parent?: TraceSpan): Promise<WatchdogSummar
6599
7675
  // one (the cron path; the /debug op runs bare), ending with the action taken
6600
7676
  // — never the resource, which names a repo.
6601
7677
  const checkOne = async (record: { resource: string }, span?: TraceSpan) => {
6602
- const check = await residentStub(env, record.resource).watchdogCheck();
7678
+ const stub = residentStub(env, record.resource);
7679
+ const check = await stub.watchdogCheck();
6603
7680
  if (check.action === "provision-timed-out") {
6604
7681
  // The DO already tried to release its own slot; this is the backstop.
6605
7682
  await registry.remove(record.resource);
6606
7683
  }
7684
+ const instance = await createRefreshInstance(env, stub, record.resource, check.refresh);
6607
7685
  span?.setAttrs({ outcome: check.action });
6608
- return check;
7686
+ return { ...check, instance };
6609
7687
  };
6610
7688
  const settled = await Promise.allSettled(
6611
7689
  residents.map((record) =>
@@ -6621,12 +7699,141 @@ async function runWatchdog(env: Env, parent?: TraceSpan): Promise<WatchdogSummar
6621
7699
  reason: s.value.reason,
6622
7700
  action: s.value.action,
6623
7701
  disk: s.value.disk,
7702
+ lifecycle: s.value.refresh.lifecycle,
7703
+ instance: s.value.instance,
6624
7704
  }
6625
7705
  : { resource: record.resource, error: errMsg(s.reason) };
6626
7706
  });
6627
7707
  return { cap: (await registry.limits()).cap, count: residents.length, results };
6628
7708
  }
6629
7709
 
7710
+ // ---------------------------------------------------------------------------
7711
+ // The refresh cycle as a Workflow instance (docs/reference/specs/resident-repos.md item 7)
7712
+ // ---------------------------------------------------------------------------
7713
+
7714
+ /** What an instance answers when it ends: small facts for the engine's record. */
7715
+ interface RefreshInstanceSummary {
7716
+ instance: string;
7717
+ /** `ok`, or the word a gate or a failure ended the cycle with. */
7718
+ outcome: string;
7719
+ step: "fetch" | "install" | "build" | "snapshot";
7720
+ action?: RefreshPlan["action"];
7721
+ sha?: string;
7722
+ }
7723
+
7724
+ /** One refresh cycle as one short Workflow instance: `fetch`, `install` (only
7725
+ * when the plan moved the lockfile key), `build` (only when the branch
7726
+ * moved), `snapshot` — each a `step.do` calling the resident's own step
7727
+ * method through the DO stub, under the retry policy `REFRESH_STEP_RETRIES`
7728
+ * (six attempts, thirty seconds apart, doubling: about 15.5 minutes, past
7729
+ * the 3 to 10 minutes a resident Worker rollover takes to settle) and a
7730
+ * timeout equal to the method's own budget (never above the engine's 30
7731
+ * minutes). Inputs to a step are the event's `resource`, the instance id and
7732
+ * previous steps' returns — refs, shas, a key, a path — never a payload and
7733
+ * never a credential. A step killed from outside (the container replaced
7734
+ * under it) throws and the engine retries it into the same idempotent
7735
+ * method; a gate that ends the cycle (idle, a container restart) or a
7736
+ * failure of the repository's own (recorded as `degraded`, the last snapshot
7737
+ * still serving) ends the instance with that word, and the next cron firing
7738
+ * creates the next one from the row's state. The instance runs one cycle
7739
+ * and returns: it is created by the watchdog cron per resident and
7740
+ * ten-minute bucket (`createRefreshInstance`), never a loop.
7741
+ *
7742
+ * The run is the cycle's root span, `resident.refresh` carrying the instance
7743
+ * id (docs/reference/specs/tracing.md item 25), with every command a step ran
7744
+ * grafted under it as a `resident.<step>` child — the same shape the alarm's
7745
+ * root has. */
7746
+ export class ResidentRefresh extends WorkflowEntrypoint<Env, RefreshInstanceParams> {
7747
+ async run(
7748
+ event: Readonly<WorkflowEvent<RefreshInstanceParams>>,
7749
+ step: WorkflowStep,
7750
+ ): Promise<RefreshInstanceSummary> {
7751
+ const { resource } = event.payload;
7752
+ const instance = event.instanceId;
7753
+ const stub = residentStub(this.env, resource);
7754
+ const root = startAdoptedRoot(tracer, "resident.refresh", {
7755
+ sinks: traceSinks,
7756
+ startedAt: event.timestamp.getTime(),
7757
+ attrs: { instanceId: instance },
7758
+ });
7759
+ const graft = (answer: InstanceStepTrace) =>
7760
+ graftResidentSteps(answer.trace, {
7761
+ parent: root,
7762
+ prefix: "resident",
7763
+ baseAt: answer.startedAt,
7764
+ clipAt: systemClock(),
7765
+ });
7766
+ const retries = REFRESH_STEP_RETRIES;
7767
+ /** The word a step that did not finish ends the instance with, and its outcome for the root. */
7768
+ const ended = (
7769
+ at: RefreshInstanceSummary["step"],
7770
+ answer: { status: "stopped"; why: string } | { status: "failed"; reason: string },
7771
+ ): RefreshInstanceSummary => ({
7772
+ instance,
7773
+ outcome: answer.status === "stopped" ? answer.why : "failed",
7774
+ step: at,
7775
+ });
7776
+ let summary: RefreshInstanceSummary | undefined;
7777
+ try {
7778
+ const fetched = await step.do("fetch", { retries, timeout: stepTimeoutMs(REFRESH_FETCH_STEP_BUDGET_MS) }, () =>
7779
+ stub.refreshInstanceFetch({ resource, instance }),
7780
+ );
7781
+ graft(fetched);
7782
+ if (fetched.status !== "done") return (summary = ended("fetch", fetched));
7783
+ let depsEntry: string | null = null;
7784
+ if (fetched.install) {
7785
+ const installed = await step.do(
7786
+ "install",
7787
+ { retries, timeout: stepTimeoutMs(REFRESH_INSTALL_STEP_BUDGET_MS) },
7788
+ () => stub.refreshInstanceInstall({ resource, instance, sha: fetched.sha, lockfileKey: fetched.lockfileKey }),
7789
+ );
7790
+ graft(installed);
7791
+ if (installed.status !== "done") return (summary = ended("install", installed));
7792
+ depsEntry = installed.entry;
7793
+ }
7794
+ if (fetched.action !== "unchanged") {
7795
+ const built = await step.do("build", { retries, timeout: stepTimeoutMs(REFRESH_BUILD_STEP_BUDGET_MS) }, () =>
7796
+ stub.refreshInstanceBuild({
7797
+ resource,
7798
+ instance,
7799
+ sha: fetched.sha,
7800
+ factsSha: fetched.factsSha,
7801
+ lockfileKey: fetched.lockfileKey,
7802
+ depsEntry,
7803
+ }),
7804
+ );
7805
+ graft(built);
7806
+ if (built.status !== "done") return (summary = ended("build", built));
7807
+ }
7808
+ const snapped = await step.do(
7809
+ "snapshot",
7810
+ { retries, timeout: stepTimeoutMs(REFRESH_SNAPSHOT_STEP_BUDGET_MS) },
7811
+ () =>
7812
+ stub.refreshInstanceSnapshot({
7813
+ resource,
7814
+ instance,
7815
+ ref: fetched.ref,
7816
+ sha: fetched.sha,
7817
+ lockfileKey: fetched.lockfileKey,
7818
+ action: fetched.action,
7819
+ mintError: fetched.mintError,
7820
+ }),
7821
+ );
7822
+ graft(snapped);
7823
+ if (snapped.status !== "done") return (summary = ended("snapshot", snapped));
7824
+ return (summary = { instance, outcome: "ok", step: "snapshot", action: fetched.action, sha: fetched.sha });
7825
+ } catch (err) {
7826
+ // A step out of retries: the engine records the failed instance by id;
7827
+ // the row keeps its last state and the next cron firing starts the next cycle.
7828
+ root.fail(err);
7829
+ throw err;
7830
+ } finally {
7831
+ const outcome = summary?.outcome ?? "error";
7832
+ root.end(outcome === "ok" || (summary !== undefined && outcome !== "failed") ? "ok" : "error", { outcome });
7833
+ }
7834
+ }
7835
+ }
7836
+
6630
7837
  function json(data: unknown, status = 200): Response {
6631
7838
  // Item 62: every non-streamed body leaves through here; its `error`,
6632
7839
  // `reason` and `summary` strings are made safe at the exit.