@coreplane/switchboard 1.19.3 → 1.200.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/config/config.example.yaml +9 -0
- package/dist/assets/deploy/cloudflare/package.json +1 -1
- package/dist/assets/deploy/cloudflare-resident/package.json +1 -1
- package/dist/assets/deploy/cloudflare-resident/worker.ts +1663 -456
- package/dist/assets/deploy/cloudflare-resident/wrangler.template.jsonc +12 -0
- package/dist/assets/deploy/cloudflare-sandbox/package.json +1 -1
- package/dist/assets/package-lock.json +11 -22
- package/dist/assets/package.json +1 -1
- package/dist/assets/project.json +1 -1
- package/dist/assets/source.json +3 -3
- package/dist/assets/src/core/schedules.ts +7 -2
- package/dist/assets/src/core/trace/attrs.ts +7 -0
- package/dist/assets/src/execution/residentDepsStore.ts +16 -0
- package/dist/assets/src/execution/residentIncarnation.ts +134 -0
- package/dist/assets/src/execution/residentInstanceId.ts +177 -0
- package/dist/assets/src/execution/residentStepPlan.ts +128 -0
- package/dist/assets/web/dist/.vite/manifest.json +18 -18
- package/dist/assets/web/dist/assets/CostsPage-CzutIHUw.js +1 -0
- package/dist/assets/web/dist/assets/{ResidentDetailPage-DvQ05AGa.js → ResidentDetailPage-Dw3e9u1u.js} +1 -1
- package/dist/assets/web/dist/assets/{ResidentsIndexPage-B3uxKUne.js → ResidentsIndexPage-D8NN4AFS.js} +1 -1
- package/dist/assets/web/dist/assets/{RunRoutePage-ty94olNM.js → RunRoutePage-C1TGcPCl.js} +1 -1
- package/dist/assets/web/dist/assets/{RunsIndexPage-CM-qxyQm.js → RunsIndexPage-D9iLpfAr.js} +1 -1
- package/dist/assets/web/dist/assets/{ScheduledPage-C1psvLD4.js → ScheduledPage-BJCtMp7H.js} +1 -1
- package/dist/assets/web/dist/assets/{StatusDot-DuoQnQeU.js → StatusDot-DvV0z_-s.js} +1 -1
- package/dist/assets/web/dist/assets/{Tooltip-BfLPyxQy.js → Tooltip-C7yR095Y.js} +1 -1
- package/dist/assets/web/dist/assets/{main-Bnbk_Rsg.js → main-DDu3Ig6S.js} +2 -2
- package/dist/cli.js +66 -9
- package/package.json +1 -1
- package/dist/assets/web/dist/assets/CostsPage-CTZcMYYx.js +0 -1
|
@@ -77,7 +77,7 @@ import {
|
|
|
77
77
|
import { AsyncLocalStorage } from "node:async_hooks";
|
|
78
78
|
import type { DirectoryBackup, SandboxCommand } from "@cloudflare/sandbox";
|
|
79
79
|
import { createExtensionProcessSandbox } from "@cloudflare/sandbox/extensions";
|
|
80
|
-
import { DurableObject } from "cloudflare:workers";
|
|
80
|
+
import { DurableObject, WorkflowEntrypoint, type WorkflowEvent, type WorkflowStep } from "cloudflare:workers";
|
|
81
81
|
import { BASH_TIMEOUT_MAX_MS, BASH_TIMEOUT_MS, clampBashTimeout } from "../../src/execution/bashTimeout.js";
|
|
82
82
|
import { selectBindingsToPurge } from "../../src/execution/bindingPurge.js";
|
|
83
83
|
import { busyAfterKillReason, planForceDetach } from "../../src/execution/residentDetach.js";
|
|
@@ -141,10 +141,47 @@ import {
|
|
|
141
141
|
restoreArchivePath,
|
|
142
142
|
withTimeout,
|
|
143
143
|
type RefreshDisk,
|
|
144
|
+
type RefreshPlan,
|
|
144
145
|
type RestoreSample,
|
|
145
146
|
type RefreshFailure,
|
|
146
147
|
type RefreshOutcome,
|
|
147
148
|
} from "../../src/execution/residentRefresh.js";
|
|
149
|
+
import {
|
|
150
|
+
REFRESH_STEP_RETRIES,
|
|
151
|
+
lifecycleOf,
|
|
152
|
+
parseLifecycle,
|
|
153
|
+
refreshInstanceId,
|
|
154
|
+
shouldCreateRefreshInstance,
|
|
155
|
+
stepTimeoutMs,
|
|
156
|
+
type RefreshRow,
|
|
157
|
+
type ResidentLifecycle,
|
|
158
|
+
} from "../../src/execution/residentInstanceId.js";
|
|
159
|
+
import {
|
|
160
|
+
MIRROR_MUTEX_KEY,
|
|
161
|
+
REFRESH_CYCLE_LEASE_MS,
|
|
162
|
+
STALE_MIDFLIGHT_MS,
|
|
163
|
+
depsLeaseKey,
|
|
164
|
+
liveInFlight,
|
|
165
|
+
mintIncarnationId,
|
|
166
|
+
inFlightKey,
|
|
167
|
+
inFlightRow,
|
|
168
|
+
releaseMutex,
|
|
169
|
+
takeMutex,
|
|
170
|
+
type InFlightRow,
|
|
171
|
+
type Lease,
|
|
172
|
+
} from "../../src/execution/residentIncarnation.js";
|
|
173
|
+
import {
|
|
174
|
+
LAST_FETCH_KEY,
|
|
175
|
+
planBuild,
|
|
176
|
+
planFetchMirror,
|
|
177
|
+
planInstallDeps,
|
|
178
|
+
planMaterializeDeps,
|
|
179
|
+
planRestore,
|
|
180
|
+
planSnapshot,
|
|
181
|
+
snapshotCommitDecision,
|
|
182
|
+
type FetchRecord,
|
|
183
|
+
type SnapshotStamp,
|
|
184
|
+
} from "../../src/execution/residentStepPlan.js";
|
|
148
185
|
import {
|
|
149
186
|
DF_FREE_ARGV,
|
|
150
187
|
DISK_FULL_FREE_KIB,
|
|
@@ -204,6 +241,8 @@ import {
|
|
|
204
241
|
depsEntryPath,
|
|
205
242
|
depsInstallSemaphoreSize,
|
|
206
243
|
depsScratchCloneArgv,
|
|
244
|
+
depsAttemptOfScratchPath,
|
|
245
|
+
depsAttemptPaths,
|
|
207
246
|
depsScratchPath,
|
|
208
247
|
depsStagingPath,
|
|
209
248
|
depsHardenScript,
|
|
@@ -257,10 +296,21 @@ function refusalOutcome(err: ThreadErr): string {
|
|
|
257
296
|
return err.needs ? `needs_${err.needs}` : "error";
|
|
258
297
|
}
|
|
259
298
|
|
|
299
|
+
/** What a refresh instance is created with: the resident it runs for. Every
|
|
300
|
+
* other input is read from the resident's rows at each step, never carried. */
|
|
301
|
+
interface RefreshInstanceParams {
|
|
302
|
+
resource: string;
|
|
303
|
+
}
|
|
304
|
+
|
|
260
305
|
interface Env {
|
|
261
306
|
RESIDENT: DurableObjectNamespace<ResidentDO>;
|
|
262
307
|
REGISTRY: DurableObjectNamespace<ResidentRegistryDO>;
|
|
263
308
|
BACKUP_BUCKET: R2Bucket;
|
|
309
|
+
/** The refresh cycle as a Workflow instance (docs/reference/specs/resident-repos.md
|
|
310
|
+
* item 7): `ResidentRefresh` below. The watchdog cron creates one per
|
|
311
|
+
* resident whose row says `lifecycle: workflow`; an `alarm` resident (the
|
|
312
|
+
* default) never has one. */
|
|
313
|
+
RESIDENT_REFRESH: Workflow<RefreshInstanceParams>;
|
|
264
314
|
// Presigned snapshot transfers (docs/reference/specs/resident-repos.md item 61): with all
|
|
265
315
|
// four present the container moves archive bytes itself over presigned R2
|
|
266
316
|
// URLs and the DO stays out of the data path; any one absent → the SDK's
|
|
@@ -358,6 +408,19 @@ const KILL_EXIT_WAIT_MS = 10_000;
|
|
|
358
408
|
* Same class as the other network budgets (observed live transfers run
|
|
359
409
|
* seconds, recorded in `lastRestore.ms`). */
|
|
360
410
|
const R2_TRANSFER_TIMEOUT_MS = 5 * 60_000;
|
|
411
|
+
/** The mirror-mutex lease for a section that names no step budget of its own
|
|
412
|
+
* (attach's clone section, a sweep's eviction, the wake and reclaim fetches):
|
|
413
|
+
* a holder of the current incarnation still holding past this has hung, the
|
|
414
|
+
* same bound the watchdog puts on a mid-flight state. The engine steps pass
|
|
415
|
+
* their exact budgets instead. */
|
|
416
|
+
const MIRROR_LEASE_DEFAULT_MS = STALE_MIDFLIGHT_MS;
|
|
417
|
+
/** What the dependency install step runs around the install itself, each
|
|
418
|
+
* bounded: the scratch clone, the seed and its cache swap, the commit (one
|
|
419
|
+
* network budget each) and the harden (the default exec budget). The step's
|
|
420
|
+
* lease is the install budget plus this. */
|
|
421
|
+
const DEPS_STEP_OVERHEAD_MS = 4 * GIT_NETWORK_TIMEOUT_MS + DEFAULT_EXEC_TIMEOUT_MS;
|
|
422
|
+
|
|
423
|
+
const sleep = (ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms));
|
|
361
424
|
|
|
362
425
|
/** On-disk layout inside the resident container (disk is cache, never truth —
|
|
363
426
|
* DO storage is). Thread worktrees hang off the same mirror; keep these paths stable. */
|
|
@@ -453,10 +516,6 @@ const LRU_FLOOR_S = IDLE_AFTER_S;
|
|
|
453
516
|
/** Budget for one GitHub REST call in the reclamation pass (pulls lookup per
|
|
454
517
|
* live non-default binding); a slow API answers "unknown", never blocks the cycle. */
|
|
455
518
|
const GITHUB_API_TIMEOUT_MS = 10_000;
|
|
456
|
-
/** A `refreshing`/`restoring` marker older than this with nothing running is
|
|
457
|
-
* an orphan from an interrupted cycle; the watchdog normalizes it. Comfortably
|
|
458
|
-
* above the longest legitimate cycle (REFRESH_BUILD_TIMEOUT_MS-scale installs). */
|
|
459
|
-
const STALE_MIDFLIGHT_MS = 30 * 60_000;
|
|
460
519
|
/** A resident degraded with the SAME reason for this many consecutive cycles
|
|
461
520
|
* is chronically broken (e.g. the default branch's build fails); retrying
|
|
462
521
|
* every 10 min bills the container 24/7 for nothing. After the streak it may
|
|
@@ -493,6 +552,24 @@ const DISK_FULL_REARM_S = 1;
|
|
|
493
552
|
* detach and sweep eviction. Surfaced as the live view's `disk`; the attach
|
|
494
553
|
* admission projects a new tree's cost from its parts. */
|
|
495
554
|
const DISK_KEY = "resident:disk";
|
|
555
|
+
/** Which scheduler drives this resident's refresh cycle (docs/reference/specs/resident-repos.md
|
|
556
|
+
* item 7): the alarm chain (the default — a row without the key reads `alarm`)
|
|
557
|
+
* or the Workflow instance the watchdog cron creates. Set through the admin
|
|
558
|
+
* `/debug` `lifecycle` op; read by the alarm's entry, the watchdog's re-arm
|
|
559
|
+
* branches and the cron's instance-creation duty, so the two schedulers never
|
|
560
|
+
* both drive a cycle for one resident. */
|
|
561
|
+
const LIFECYCLE_KEY = "resident:lifecycle";
|
|
562
|
+
/** The refresh instance row (item 7): the last instance the cron created for
|
|
563
|
+
* this resident, with the step it last reported and the cycle lease it holds,
|
|
564
|
+
* and the last bucket the cron skipped (a live cycle, a duplicate id). */
|
|
565
|
+
const REFRESH_INSTANCE_KEY = "resident:refreshInstance";
|
|
566
|
+
/** The refresh instance's step budgets: each `step.do` timeout is the DO
|
|
567
|
+
* method's own budget, capped at the engine's 30-minute step ceiling
|
|
568
|
+
* (`stepTimeoutMs`), so a step timeout and a command timeout agree. */
|
|
569
|
+
const REFRESH_FETCH_STEP_BUDGET_MS = RESTORE_MAX_MS + GIT_NETWORK_TIMEOUT_MS; // a wake's restore, then the fetch
|
|
570
|
+
const REFRESH_INSTALL_STEP_BUDGET_MS = REFRESH_INSTALL_TIMEOUT_MS + DEPS_STEP_OVERHEAD_MS; // the install's own lease
|
|
571
|
+
const REFRESH_BUILD_STEP_BUDGET_MS = GIT_NETWORK_TIMEOUT_MS + REFRESH_BUILD_TIMEOUT_MS; // the build's mutex lease
|
|
572
|
+
const REFRESH_SNAPSHOT_STEP_BUDGET_MS = R2_TRANSFER_TIMEOUT_MS + GIT_NETWORK_TIMEOUT_MS; // the archives, then the reclaim pass and the disk sample
|
|
496
573
|
const DISK_MEASURE_CALLBACK = "onDiskMeasure";
|
|
497
574
|
const DISK_MEASURE_DELAY_S = 1;
|
|
498
575
|
/** A `du` over a multi-GB checkout plus every live tree is seconds warm, tens
|
|
@@ -1028,6 +1105,61 @@ interface DepsBackupRecord {
|
|
|
1028
1105
|
createdAt: string;
|
|
1029
1106
|
}
|
|
1030
1107
|
|
|
1108
|
+
/** What the snapshot step answers: the record already at the stamp, a record
|
|
1109
|
+
* it committed (with the one it replaced), or a step another writer won. */
|
|
1110
|
+
type SnapshotStepResult =
|
|
1111
|
+
| { done: true; record: SnapshotRecord }
|
|
1112
|
+
| { done: false; superseded: false; record: SnapshotRecord; previous: SnapshotRecord | undefined }
|
|
1113
|
+
| { done: false; superseded: true };
|
|
1114
|
+
|
|
1115
|
+
/** The refresh instance row (REFRESH_INSTANCE_KEY, item 7). */
|
|
1116
|
+
interface RefreshInstanceRow {
|
|
1117
|
+
/** The most recent instance the cron created (or a step reported) for this resident. */
|
|
1118
|
+
instance: {
|
|
1119
|
+
id: string;
|
|
1120
|
+
createdAt: string;
|
|
1121
|
+
/** `<step>: <outcome>` of the step the instance last ran; null before its first. */
|
|
1122
|
+
lastStep: string | null;
|
|
1123
|
+
/** The cycle lease the instance holds in the in-flight row, from its fetch step to its end. */
|
|
1124
|
+
holder: string | null;
|
|
1125
|
+
} | null;
|
|
1126
|
+
/** The most recent bucket the cron did not create for: a live cycle, or the engine's duplicate-id refusal. */
|
|
1127
|
+
skipped: { id: string; at: string; why: string } | null;
|
|
1128
|
+
}
|
|
1129
|
+
|
|
1130
|
+
/** What every instance step answers besides its own facts: the resident's
|
|
1131
|
+
* wall clock at the step's start and the commands it ran, so the instance
|
|
1132
|
+
* can graft them under its root the way the bot grafts an attach's. */
|
|
1133
|
+
interface InstanceStepTrace {
|
|
1134
|
+
startedAt: number;
|
|
1135
|
+
trace: ResidentStep[];
|
|
1136
|
+
}
|
|
1137
|
+
/** A step's own verdict: `done` with its facts; `stopped` by a gate that ends
|
|
1138
|
+
* the cycle with nothing to record (idle, a container restart, an offboard);
|
|
1139
|
+
* `failed` by the repository's own doing, already recorded as `degraded`.
|
|
1140
|
+
* A step killed from outside answers none of these — it throws, and the
|
|
1141
|
+
* engine retries it. */
|
|
1142
|
+
type InstanceStepResult<T> =
|
|
1143
|
+
({ status: "done" } & T) | { status: "stopped"; why: string } | { status: "failed"; reason: string };
|
|
1144
|
+
type InstanceStepAnswer<T> = InstanceStepResult<T> & InstanceStepTrace;
|
|
1145
|
+
/** The fetch step's facts, small by construction: refs, shas, keys, words. */
|
|
1146
|
+
interface RefreshFetchFacts {
|
|
1147
|
+
ref: string;
|
|
1148
|
+
sha: string;
|
|
1149
|
+
factsSha: string;
|
|
1150
|
+
lockfileKey: string;
|
|
1151
|
+
action: RefreshPlan["action"];
|
|
1152
|
+
/** Whether the install step must run: a rebuild whose lockfile key moved, on a repo with an install command. */
|
|
1153
|
+
install: boolean;
|
|
1154
|
+
mintError: string | null;
|
|
1155
|
+
}
|
|
1156
|
+
/** What the cron did about one resident's refresh instance this pass. */
|
|
1157
|
+
interface RefreshInstanceAction {
|
|
1158
|
+
id: string;
|
|
1159
|
+
action: "created" | "duplicate" | "skipped" | "failed";
|
|
1160
|
+
why: string;
|
|
1161
|
+
}
|
|
1162
|
+
|
|
1031
1163
|
// ---------------------------------------------------------------------------
|
|
1032
1164
|
// Registry DO (singleton): onboarded set + config, atomic cap enforcement
|
|
1033
1165
|
// ---------------------------------------------------------------------------
|
|
@@ -1385,8 +1517,9 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
1385
1517
|
// because every way the fact can stop being true is observable and clears
|
|
1386
1518
|
// the memos: a runtime replacement surfaces as RuntimeReplacedError at the
|
|
1387
1519
|
// ONE exec choke point (`run()`), a deliberate stop/teardown/rebuild calls
|
|
1388
|
-
// `
|
|
1389
|
-
// (`setResidentState`) clears
|
|
1520
|
+
// `swapIncarnation()` at its site (memos AND the incarnation id go), every
|
|
1521
|
+
// lifecycle transition (`setResidentState`) clears the memos alone — the
|
|
1522
|
+
// incarnation survives it — and a sleep cannot race the TTL — the
|
|
1390
1523
|
// container sleeps only after SLEEP_AFTER (20 min) of idleness, while the
|
|
1391
1524
|
// hydration memo lives `hydrationMemoTtlMs` (60 s) past the last activity
|
|
1392
1525
|
// that set it. Storage stays the truth: the memo caches a verdict PROBED from
|
|
@@ -1396,6 +1529,29 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
1396
1529
|
private gitSetupDone = false;
|
|
1397
1530
|
private stageDirsReady = new Set<string>();
|
|
1398
1531
|
|
|
1532
|
+
/** The incarnation: one isolate paired with one container runtime
|
|
1533
|
+
* (docs/reference/specs/resident-repos.md item 22). Minted when the object
|
|
1534
|
+
* starts and again ONLY when the runtime is replaced under it
|
|
1535
|
+
* (`swapIncarnation`) — every lease this object writes carries it, and a
|
|
1536
|
+
* lease from another incarnation is a holder whose process context is gone.
|
|
1537
|
+
* A lifecycle transition is not a swap: the cycle that flips the state to
|
|
1538
|
+
* `refreshing` holds a lease of this incarnation and must still hold it
|
|
1539
|
+
* afterwards, so `setResidentState` clears the memos and nothing more. */
|
|
1540
|
+
private incarnation = mintIncarnationId();
|
|
1541
|
+
private leaseSeq = 0;
|
|
1542
|
+
|
|
1543
|
+
private nextHolder(): string {
|
|
1544
|
+
return `${this.incarnation}:${++this.leaseSeq}`;
|
|
1545
|
+
}
|
|
1546
|
+
|
|
1547
|
+
/** The runtime under this object is gone (a replacement seen at the exec
|
|
1548
|
+
* choke point, a deliberate stop/teardown/rebuild, retirement): every lease
|
|
1549
|
+
* this incarnation holds is dead from here on, and so are its memos. */
|
|
1550
|
+
private swapIncarnation(): void {
|
|
1551
|
+
this.incarnation = mintIncarnationId();
|
|
1552
|
+
this.clearIncarnationMemos();
|
|
1553
|
+
}
|
|
1554
|
+
|
|
1399
1555
|
private clearIncarnationMemos(): void {
|
|
1400
1556
|
this.hydratedVerdictAt = 0;
|
|
1401
1557
|
this.gitSetupDone = false;
|
|
@@ -1404,18 +1560,25 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
1404
1560
|
this.depsInstallSlots = null;
|
|
1405
1561
|
}
|
|
1406
1562
|
|
|
1407
|
-
/** Mirror mutex: a DO yields at every await, so two
|
|
1408
|
-
* requests CAN interleave mid-handler — every mirror mutation
|
|
1409
|
-
* worktree add/remove)
|
|
1410
|
-
*
|
|
1411
|
-
*
|
|
1412
|
-
*
|
|
1563
|
+
/** Mirror mutex, in-process half: a DO yields at every await, so two
|
|
1564
|
+
* in-flight requests CAN interleave mid-handler — every mirror mutation
|
|
1565
|
+
* (fetch, worktree add/remove) queues on this promise chain, in arrival
|
|
1566
|
+
* order. The chain is the fast path within one incarnation; the durable
|
|
1567
|
+
* row (MIRROR_MUTEX_KEY) is the truth across incarnations: an isolate swap
|
|
1568
|
+
* drops the chain while the holder's process may keep writing, and the row
|
|
1569
|
+
* is what the next incarnation reads before it touches the tree. */
|
|
1413
1570
|
private mirrorLockTail: Promise<void> = Promise.resolve();
|
|
1414
1571
|
|
|
1415
1572
|
/** Run `fn` holding the mirror mutex. With waitTimeoutMs > 0, gives up
|
|
1416
1573
|
* waiting after that long (throws MirrorBusyError) — the queued slot is
|
|
1417
|
-
* released so later waiters are not stuck behind a ghost.
|
|
1418
|
-
|
|
1574
|
+
* released so later waiters are not stuck behind a ghost. `lease` names
|
|
1575
|
+
* the step and its budget on the row; a section without a budget of its
|
|
1576
|
+
* own gets the default lease. */
|
|
1577
|
+
private async withMirrorLock<T>(
|
|
1578
|
+
fn: () => Promise<T>,
|
|
1579
|
+
waitTimeoutMs = 0,
|
|
1580
|
+
lease: { step: string; budgetMs: number } = { step: "mirror", budgetMs: MIRROR_LEASE_DEFAULT_MS },
|
|
1581
|
+
): Promise<{ value: T; waitedMs: number }> {
|
|
1419
1582
|
const prev = this.mirrorLockTail;
|
|
1420
1583
|
let release!: () => void;
|
|
1421
1584
|
const slot = new Promise<void>((resolve) => (release = resolve));
|
|
@@ -1445,15 +1608,88 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
1445
1608
|
} else {
|
|
1446
1609
|
await prev.catch(() => {});
|
|
1447
1610
|
}
|
|
1611
|
+
// Past the chain, the row: taken when free or when its holder is dead.
|
|
1612
|
+
const holder = this.nextHolder();
|
|
1613
|
+
try {
|
|
1614
|
+
await this.takeMirrorRow(holder, lease, waitTimeoutMs, started);
|
|
1615
|
+
} catch (err) {
|
|
1616
|
+
release();
|
|
1617
|
+
throw err;
|
|
1618
|
+
}
|
|
1448
1619
|
const waitedMs = systemClock() - started;
|
|
1449
1620
|
this.stepTrace.getStore()?.mutexWait(waitedMs, systemClock());
|
|
1450
1621
|
try {
|
|
1451
1622
|
return { value: await fn(), waitedMs };
|
|
1452
1623
|
} finally {
|
|
1453
|
-
|
|
1624
|
+
try {
|
|
1625
|
+
// A failed release must not replace fn's result: the row's expiry is
|
|
1626
|
+
// the backstop, and the next taker takes over a dead holder anyway.
|
|
1627
|
+
await this.releaseLease(MIRROR_MUTEX_KEY, holder).catch((err: unknown) => {
|
|
1628
|
+
console.log(`mirror mutex: release of ${holder} failed (${errMsg(err)}); the lease expires on its own`);
|
|
1629
|
+
});
|
|
1630
|
+
} finally {
|
|
1631
|
+
release();
|
|
1632
|
+
}
|
|
1633
|
+
}
|
|
1634
|
+
}
|
|
1635
|
+
|
|
1636
|
+
/** Write the mirror-mutex row for `holder`, or wait for a live holder of
|
|
1637
|
+
* this incarnation to end. A dead holder — another incarnation, or one past
|
|
1638
|
+
* its budget — is taken over at once (takeMutex); the row only ever names a
|
|
1639
|
+
* live one of THIS incarnation when a release is racing this read, so the
|
|
1640
|
+
* wait is short and bounded by the caller's timeout like the chain wait. */
|
|
1641
|
+
private async takeMirrorRow(
|
|
1642
|
+
holder: string,
|
|
1643
|
+
lease: { step: string; budgetMs: number },
|
|
1644
|
+
waitTimeoutMs: number,
|
|
1645
|
+
started: number,
|
|
1646
|
+
): Promise<void> {
|
|
1647
|
+
for (;;) {
|
|
1648
|
+
const row = await this.ctx.storage.get<Lease>(MIRROR_MUTEX_KEY);
|
|
1649
|
+
const decision = takeMutex(row, systemClock(), this.incarnation, lease.budgetMs, lease.step, holder);
|
|
1650
|
+
if (decision.action === "take") {
|
|
1651
|
+
if (decision.dead) {
|
|
1652
|
+
console.log(
|
|
1653
|
+
`mirror mutex: ${decision.why} — ${lease.step} takes over from ${decision.dead.step} (${decision.dead.holder})`,
|
|
1654
|
+
);
|
|
1655
|
+
}
|
|
1656
|
+
await this.ctx.storage.put(MIRROR_MUTEX_KEY, decision.row);
|
|
1657
|
+
return;
|
|
1658
|
+
}
|
|
1659
|
+
const left = waitTimeoutMs > 0 ? waitTimeoutMs - (systemClock() - started) : Infinity;
|
|
1660
|
+
if (left <= 0) throw new MirrorBusyError(`mirror-busy: mutex not acquired within ${waitTimeoutMs}ms`);
|
|
1661
|
+
await sleep(Math.min(decision.remainingMs + 1, left, 1_000));
|
|
1454
1662
|
}
|
|
1455
1663
|
}
|
|
1456
1664
|
|
|
1665
|
+
/** Delete a lease row iff `holder` still owns it (releaseMutex). */
|
|
1666
|
+
private async releaseLease(key: string, holder: string): Promise<void> {
|
|
1667
|
+
const outcome = releaseMutex(await this.ctx.storage.get<Lease>(key), holder);
|
|
1668
|
+
if (outcome.released) await this.ctx.storage.delete(key);
|
|
1669
|
+
}
|
|
1670
|
+
|
|
1671
|
+
/** Lease one of the in-flight facts the watchdog reads: one document per
|
|
1672
|
+
* fact (inFlightKey), a plain put, so a hydration's clear can never race a
|
|
1673
|
+
* refresh's record on a shared row. */
|
|
1674
|
+
private async recordInFlight(kind: keyof InFlightRow, holder: string, budgetMs: number, step: string): Promise<void> {
|
|
1675
|
+
const lease: Lease = { holder, incarnation: this.incarnation, expiresAt: systemClock() + budgetMs, step };
|
|
1676
|
+
await this.ctx.storage.put(inFlightKey(kind), lease);
|
|
1677
|
+
}
|
|
1678
|
+
|
|
1679
|
+
private async clearInFlight(kind: keyof InFlightRow, holder: string): Promise<void> {
|
|
1680
|
+
const lease = await this.ctx.storage.get<Lease>(inFlightKey(kind));
|
|
1681
|
+
if (lease?.holder === holder) await this.ctx.storage.delete(inFlightKey(kind));
|
|
1682
|
+
}
|
|
1683
|
+
|
|
1684
|
+
/** The two in-flight facts as one row, for liveInFlight and /debug. */
|
|
1685
|
+
private async readInFlight(): Promise<InFlightRow> {
|
|
1686
|
+
const [refresh, hydration] = await Promise.all([
|
|
1687
|
+
this.ctx.storage.get<Lease>(inFlightKey("refresh")),
|
|
1688
|
+
this.ctx.storage.get<Lease>(inFlightKey("hydration")),
|
|
1689
|
+
]);
|
|
1690
|
+
return inFlightRow(refresh, hydration);
|
|
1691
|
+
}
|
|
1692
|
+
|
|
1457
1693
|
private registry() {
|
|
1458
1694
|
return this.env.REGISTRY.get(this.env.REGISTRY.idFromName("registry"));
|
|
1459
1695
|
}
|
|
@@ -1491,7 +1727,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
1491
1727
|
// takes the throw below. It exists so that if a future SDK vouches "never
|
|
1492
1728
|
// started" we retry then — and only then — without a change here.
|
|
1493
1729
|
if (!(err instanceof OperationInterruptedError && err.retryable === true)) {
|
|
1494
|
-
this.
|
|
1730
|
+
this.swapIncarnation(); // the container this incarnation's memos described is gone
|
|
1495
1731
|
throw new RuntimeReplacedError("spawn", err);
|
|
1496
1732
|
}
|
|
1497
1733
|
console.log(
|
|
@@ -1504,7 +1740,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
1504
1740
|
return { stdout: out.stdout, stderr: out.stderr, exitCode: out.exitCode, timedOut: out.timedOut };
|
|
1505
1741
|
} catch (err) {
|
|
1506
1742
|
if (isRuntimeReplacement(err)) {
|
|
1507
|
-
this.
|
|
1743
|
+
this.swapIncarnation(); // the container this incarnation's memos described is gone
|
|
1508
1744
|
throw new RuntimeReplacedError("collect", err);
|
|
1509
1745
|
}
|
|
1510
1746
|
if (err instanceof ProcessWaitTimeoutError) {
|
|
@@ -1681,14 +1917,21 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
1681
1917
|
return (await this.runOk(["sh", "-c", script], "lockfile-key")).trim();
|
|
1682
1918
|
}
|
|
1683
1919
|
|
|
1684
|
-
/**
|
|
1685
|
-
|
|
1920
|
+
/** The sha the disk was last materialized to (READY_MARKER), or null when
|
|
1921
|
+
* either tree or the marker is missing — the restore step's fact. */
|
|
1922
|
+
private async readyStamp(): Promise<string | null> {
|
|
1686
1923
|
const r = await this.run([
|
|
1687
1924
|
"sh",
|
|
1688
1925
|
"-c",
|
|
1689
1926
|
`test -d ${MIRROR_DIR}/objects && test -d ${CHECKOUT_DIR}/.git && cat ${READY_MARKER} 2>/dev/null || echo __absent__`,
|
|
1690
1927
|
]);
|
|
1691
|
-
|
|
1928
|
+
const out = r.stdout.trim();
|
|
1929
|
+
return r.exitCode === 0 && out !== "" && out !== "__absent__" ? out : null;
|
|
1930
|
+
}
|
|
1931
|
+
|
|
1932
|
+
/** True when the disk already holds exactly what the snapshot stamp says. */
|
|
1933
|
+
private async diskMatches(sha: string): Promise<boolean> {
|
|
1934
|
+
return (await this.readyStamp()) === sha;
|
|
1692
1935
|
}
|
|
1693
1936
|
|
|
1694
1937
|
/** Write the disk markers that a materialized checkout leaves behind (see
|
|
@@ -1779,6 +2022,320 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
1779
2022
|
}
|
|
1780
2023
|
}
|
|
1781
2024
|
|
|
2025
|
+
// -- engine steps (docs/reference/specs/resident-repos.md item 22) ---------------
|
|
2026
|
+
//
|
|
2027
|
+
// Each step is one public method a cycle calls: it reads the facts it is
|
|
2028
|
+
// about to change and asks the pure plan (residentStepPlan.ts) whether the
|
|
2029
|
+
// work is done — done issues no command, so a second call with the same
|
|
2030
|
+
// inputs has no effect — then takes its lease, runs its commands under the
|
|
2031
|
+
// step's own budget, writes its result and releases. The alarm chain drives
|
|
2032
|
+
// them today in the order it always did.
|
|
2033
|
+
|
|
2034
|
+
/** Fetch the mirror from origin, once per cycle: the record under
|
|
2035
|
+
* LAST_FETCH_KEY names the cycle, so a repeated call inside the same cycle
|
|
2036
|
+
* answers the sha it already read. */
|
|
2037
|
+
async fetchMirror(input: {
|
|
2038
|
+
ref: string;
|
|
2039
|
+
cycle: string;
|
|
2040
|
+
token: string | null;
|
|
2041
|
+
}): Promise<{ done: boolean; sha: string }> {
|
|
2042
|
+
const last = await this.ctx.storage.get<FetchRecord>(LAST_FETCH_KEY);
|
|
2043
|
+
const plan = planFetchMirror({ ref: input.ref, cycle: input.cycle, last });
|
|
2044
|
+
if (plan.action === "done") return { done: true, sha: plan.sha };
|
|
2045
|
+
// The tip is read under the same lock as the fetch: an attach's own
|
|
2046
|
+
// `fetch --prune` between the two could delete the ref and fail the cycle.
|
|
2047
|
+
const { value: sha } = await this.withMirrorLock(
|
|
2048
|
+
async () => {
|
|
2049
|
+
await this.gitWithCred(
|
|
2050
|
+
input.token,
|
|
2051
|
+
["-C", MIRROR_DIR, "fetch", "--prune", "origin"],
|
|
2052
|
+
"fetch",
|
|
2053
|
+
GIT_NETWORK_TIMEOUT_MS,
|
|
2054
|
+
);
|
|
2055
|
+
return this.readMirrorSha(input.ref);
|
|
2056
|
+
},
|
|
2057
|
+
0,
|
|
2058
|
+
{ step: "fetch", budgetMs: GIT_NETWORK_TIMEOUT_MS },
|
|
2059
|
+
);
|
|
2060
|
+
await this.ctx.storage.put(LAST_FETCH_KEY, {
|
|
2061
|
+
cycle: input.cycle,
|
|
2062
|
+
ref: input.ref,
|
|
2063
|
+
sha,
|
|
2064
|
+
at: systemClock(),
|
|
2065
|
+
} satisfies FetchRecord);
|
|
2066
|
+
return { done: false, sha };
|
|
2067
|
+
}
|
|
2068
|
+
|
|
2069
|
+
/** The store entry for `key`, complete (item 59 is the primitive under it).
|
|
2070
|
+
* The install's exclusive resource is the key's entry, never the mirror —
|
|
2071
|
+
* installs run in a private scratch tree outside the mirror mutex — so its
|
|
2072
|
+
* lease is per key (depsLeaseKey). A live holder of this incarnation is the
|
|
2073
|
+
* install already running for the key: joined, never duplicated. A dead
|
|
2074
|
+
* holder left its scratch tree behind, possibly with a process still
|
|
2075
|
+
* writing into it: that tree is swept before this attempt starts, the way
|
|
2076
|
+
* every build-user step sweeps the checkout. */
|
|
2077
|
+
async installDeps(input: {
|
|
2078
|
+
key: string;
|
|
2079
|
+
sha: string;
|
|
2080
|
+
installCmd: string;
|
|
2081
|
+
budgetMs: number;
|
|
2082
|
+
seedFromKey?: string;
|
|
2083
|
+
restoreDeadlineMs?: number;
|
|
2084
|
+
}): Promise<{ done: boolean; entry: string }> {
|
|
2085
|
+
const { key } = input;
|
|
2086
|
+
const plan = planInstallDeps({ key, entryComplete: await this.depsEntryComplete(key) });
|
|
2087
|
+
if (plan.action === "done") {
|
|
2088
|
+
await this.run(["touch", depsUsedPath(key)]);
|
|
2089
|
+
return { done: true, entry: depsEntryPath(key) };
|
|
2090
|
+
}
|
|
2091
|
+
const attempt = crypto.randomUUID().slice(0, 8);
|
|
2092
|
+
const leaseKey = depsLeaseKey(key);
|
|
2093
|
+
const holder = this.nextHolder();
|
|
2094
|
+
const leaseMs = Math.max(input.budgetMs, (input.restoreDeadlineMs ?? 0) - systemClock()) + DEPS_STEP_OVERHEAD_MS;
|
|
2095
|
+
for (;;) {
|
|
2096
|
+
const row = await this.ctx.storage.get<Lease>(leaseKey);
|
|
2097
|
+
const decision = takeMutex(
|
|
2098
|
+
row,
|
|
2099
|
+
systemClock(),
|
|
2100
|
+
this.incarnation,
|
|
2101
|
+
leaseMs,
|
|
2102
|
+
"deps-install",
|
|
2103
|
+
holder,
|
|
2104
|
+
depsScratchPath(attempt),
|
|
2105
|
+
);
|
|
2106
|
+
if (decision.action === "wait") {
|
|
2107
|
+
const running = this.depsInFlight.get(key);
|
|
2108
|
+
if (running) return { done: false, entry: await running };
|
|
2109
|
+
// The lease is written before the running install registers itself;
|
|
2110
|
+
// a caller landing in between waits for that, briefly.
|
|
2111
|
+
await sleep(Math.min(decision.remainingMs + 1, 1_000));
|
|
2112
|
+
continue;
|
|
2113
|
+
}
|
|
2114
|
+
if (decision.dead?.tree) {
|
|
2115
|
+
// The dead attempt's scratch tree AND its staging dir: its install ran
|
|
2116
|
+
// in the first, its commit script was moving node_modules into the
|
|
2117
|
+
// second; a process still writing to either is killed, then both go.
|
|
2118
|
+
const deadAttempt = depsAttemptOfScratchPath(decision.dead.tree);
|
|
2119
|
+
const deadPaths = deadAttempt ? depsAttemptPaths(key, deadAttempt) : [decision.dead.tree];
|
|
2120
|
+
console.log(
|
|
2121
|
+
`deps: ${decision.why} — sweeping ${deadPaths.join(" ")} left by ${decision.dead.holder} before installing ${key.slice(0, 8)}`,
|
|
2122
|
+
);
|
|
2123
|
+
for (const dir of deadPaths) {
|
|
2124
|
+
const swept = await this.runOk(killStaleBuildProcessesCommand(BUILD_USER, dir), "deps-install-stale-sweep");
|
|
2125
|
+
if (swept.trim()) console.log(`deps: ${swept.trim()}`);
|
|
2126
|
+
}
|
|
2127
|
+
await this.run(["rm", "-rf", ...deadPaths]).catch(() => {});
|
|
2128
|
+
}
|
|
2129
|
+
await this.ctx.storage.put(leaseKey, decision.row);
|
|
2130
|
+
break;
|
|
2131
|
+
}
|
|
2132
|
+
try {
|
|
2133
|
+
const entry = await this.materializeDeps(key, input.sha, input.installCmd, input.budgetMs, {
|
|
2134
|
+
seedFromKey: input.seedFromKey,
|
|
2135
|
+
restoreDeadlineMs: input.restoreDeadlineMs,
|
|
2136
|
+
attempt,
|
|
2137
|
+
});
|
|
2138
|
+
return { done: false, entry };
|
|
2139
|
+
} finally {
|
|
2140
|
+
await this.releaseLease(leaseKey, holder);
|
|
2141
|
+
}
|
|
2142
|
+
}
|
|
2143
|
+
|
|
2144
|
+
/** Bring the checkout to `sha` and build it, under the mirror mutex. The
|
|
2145
|
+
* disk markers decide (planBuild over the refresh planner): a checkout
|
|
2146
|
+
* whose HEAD, deps and build markers all name the target is done. */
|
|
2147
|
+
async runBuild(input: {
|
|
2148
|
+
sha: string;
|
|
2149
|
+
factsSha: string;
|
|
2150
|
+
lockfileKey: string;
|
|
2151
|
+
buildCmd: string;
|
|
2152
|
+
/** The store entry to link as the checkout's node_modules when the plan installs; null when the command table has no install. */
|
|
2153
|
+
depsEntry: string | null;
|
|
2154
|
+
}): Promise<{ done: boolean; why: string }> {
|
|
2155
|
+
const { sha, lockfileKey } = input;
|
|
2156
|
+
const plan = planBuild({ sha, factsSha: input.factsSha, lockfileKey, disk: await this.readRefreshDisk() });
|
|
2157
|
+
if (plan.action === "done") return { done: true, why: plan.why };
|
|
2158
|
+
// Serialize the CHECKOUT_DIR mutation on the mirror mutex:
|
|
2159
|
+
// materializeThreadDeps reads CHECKOUT_DIR via `cp -al` under the same
|
|
2160
|
+
// lock, so an attach/op dep-copy can never hardlink a half-rebuilt
|
|
2161
|
+
// checkout into a thread tree (torn cache → false ❌ from `repo test`).
|
|
2162
|
+
// No wait timeout, exactly like the fetch lock: the background refresh
|
|
2163
|
+
// queues behind an in-flight attach instead of flipping to degraded on
|
|
2164
|
+
// transient lock contention. Token-free: repo code runs during the build.
|
|
2165
|
+
await this.withMirrorLock(
|
|
2166
|
+
async () => {
|
|
2167
|
+
// Isolation invariant (review 1b): attached, sha-pinned thread
|
|
2168
|
+
// worktrees hold hardlinks to the store entry's FILE inodes, and so
|
|
2169
|
+
// does the checkout. A build that writes THROUGH an existing inode —
|
|
2170
|
+
// many bundlers do (e.g. .next incremental manifests open+truncate
|
|
2171
|
+
// rather than recreate) — would mutate every consumer's pinned
|
|
2172
|
+
// artifacts. The `-x` clean removes the checkout's build output so
|
|
2173
|
+
// the build allocates FRESH inodes; the entry's own files are
|
|
2174
|
+
// owner-read-only (deps-harden), so a write through them fails
|
|
2175
|
+
// loudly instead of silently reaching the store; the tool caches
|
|
2176
|
+
// inside node_modules are the checkout's private copies (item 18).
|
|
2177
|
+
//
|
|
2178
|
+
// Install gate: when the committed lockfile key is unchanged,
|
|
2179
|
+
// node_modules (the view) is excluded from the clean and no deps
|
|
2180
|
+
// work happens; a changed key re-links the view to the new entry —
|
|
2181
|
+
// which is also what drops deps the new lockfile no longer has.
|
|
2182
|
+
await this.buildUserRun(checkoutUpdateCommand(sha, plan.clean), "checkout-update", GIT_NETWORK_TIMEOUT_MS);
|
|
2183
|
+
if (plan.install) {
|
|
2184
|
+
// The old view (a resumed install's keep-deps clean leaves it in
|
|
2185
|
+
// place, item 57) makes way for the new entry's: hardlinks only,
|
|
2186
|
+
// the entry's inodes are untouched.
|
|
2187
|
+
if (input.depsEntry) {
|
|
2188
|
+
await this.runOk(["rm", "-rf", `${CHECKOUT_DIR}/node_modules`], "unlink-deps-view");
|
|
2189
|
+
await this.linkDepsView(`${input.depsEntry}/node_modules`, CHECKOUT_DIR, BUILD_USER);
|
|
2190
|
+
}
|
|
2191
|
+
await this.writeDiskMarkers({ depsKey: lockfileKey });
|
|
2192
|
+
await this.runOk(["rm", "-f", INSTALLING_MARKER], "clear-installing-marker");
|
|
2193
|
+
}
|
|
2194
|
+
await this.buildUserRun(input.buildCmd, "build", REFRESH_BUILD_TIMEOUT_MS);
|
|
2195
|
+
await this.writeDiskMarkers({ builtSha: sha });
|
|
2196
|
+
},
|
|
2197
|
+
0,
|
|
2198
|
+
{ step: "build", budgetMs: GIT_NETWORK_TIMEOUT_MS + REFRESH_BUILD_TIMEOUT_MS },
|
|
2199
|
+
);
|
|
2200
|
+
return { done: false, why: plan.why };
|
|
2201
|
+
}
|
|
2202
|
+
|
|
2203
|
+
/** Archive the mirror and checkout to R2 under `stamp` and record it, with
|
|
2204
|
+
* compare-and-swap on the record read at the start: a record already at
|
|
2205
|
+
* the stamp is done; a record that moved while the archive was taken was
|
|
2206
|
+
* written by someone else and wins — the fresh objects are dropped and the
|
|
2207
|
+
* step answers `superseded`, never a throw. The recorded facts move to the
|
|
2208
|
+
* stamp in the same write, so a wake never sees a half-updated pair. */
|
|
2209
|
+
async snapshot(input: { resource: string; stamp: SnapshotStamp }): Promise<SnapshotStepResult> {
|
|
2210
|
+
const { ref, sha, lockfileHash } = input.stamp;
|
|
2211
|
+
const readAtStart = await this.ctx.storage.get<SnapshotRecord>(SNAPSHOT_KEY);
|
|
2212
|
+
const plan = planSnapshot({ stamp: input.stamp, current: readAtStart });
|
|
2213
|
+
if (plan.action === "done" && readAtStart) return { done: true, record: readAtStart };
|
|
2214
|
+
// Under the mirror mutex: nothing may mutate the checkout while it is archived.
|
|
2215
|
+
const { value: snap } = await this.withMirrorLock(
|
|
2216
|
+
() => this.takeSnapshot(input.resource, ref, sha, lockfileHash),
|
|
2217
|
+
0,
|
|
2218
|
+
{ step: "snapshot", budgetMs: R2_TRANSFER_TIMEOUT_MS },
|
|
2219
|
+
);
|
|
2220
|
+
const stored = await this.ctx.storage.get<SnapshotRecord | RepoFacts>([SNAPSHOT_KEY, FACTS_KEY]);
|
|
2221
|
+
const decision = snapshotCommitDecision({
|
|
2222
|
+
readAtStart,
|
|
2223
|
+
current: stored.get(SNAPSHOT_KEY) as SnapshotRecord | undefined,
|
|
2224
|
+
});
|
|
2225
|
+
if (decision.action === "superseded") {
|
|
2226
|
+
console.log(
|
|
2227
|
+
`snapshot: superseded — a record at ${decision.by?.sha.slice(0, 8) ?? "(none)"} moved under this step`,
|
|
2228
|
+
);
|
|
2229
|
+
await this.deleteBackupObjects([snap.mirror.id, snap.checkout.id]).catch(() => {});
|
|
2230
|
+
return { done: false, superseded: true };
|
|
2231
|
+
}
|
|
2232
|
+
const facts = stored.get(FACTS_KEY) as RepoFacts | undefined;
|
|
2233
|
+
await this.ctx.storage.put({
|
|
2234
|
+
[SNAPSHOT_KEY]: snap,
|
|
2235
|
+
...(facts
|
|
2236
|
+
? {
|
|
2237
|
+
[FACTS_KEY]: {
|
|
2238
|
+
...facts,
|
|
2239
|
+
sha,
|
|
2240
|
+
lockfileHash,
|
|
2241
|
+
lastRefreshAt: new Date(systemClock()).toISOString(),
|
|
2242
|
+
} satisfies RepoFacts,
|
|
2243
|
+
}
|
|
2244
|
+
: {}),
|
|
2245
|
+
});
|
|
2246
|
+
return { done: false, superseded: false, record: snap, previous: readAtStart };
|
|
2247
|
+
}
|
|
2248
|
+
|
|
2249
|
+
/** Bring the disk to the snapshot's stamp: the ready marker naming its sha
|
|
2250
|
+
* is done; otherwise unmount and clean, restore both archives (judged by
|
|
2251
|
+
* their bytes, against the caller's one deadline), verify the restored
|
|
2252
|
+
* mirror against the stamp and hand the checkout to the build user. Runs
|
|
2253
|
+
* inside the hydration lease its caller holds — every mirror-mutex taker
|
|
2254
|
+
* hydrates first, so nothing else touches these trees meanwhile. Failures
|
|
2255
|
+
* leave the resident `down` with the reason and the container stopped, as
|
|
2256
|
+
* the wake path always did. */
|
|
2257
|
+
async restoreCheckout(snap: SnapshotRecord, deadlineMs: number): Promise<{ done: boolean }> {
|
|
2258
|
+
const plan = planRestore({ sha: snap.sha, readyStamp: await this.readyStamp() });
|
|
2259
|
+
if (plan.action === "done") return { done: true };
|
|
2260
|
+
// A restore a previous attempt gave up on may still be writing into these
|
|
2261
|
+
// directories (the SDK call cannot be cancelled): wait for it to
|
|
2262
|
+
// settle before the clean, bounded by the same cap the restores get. A
|
|
2263
|
+
// restore that will not settle even then leaves the disk alone — a named
|
|
2264
|
+
// `down`, not a clean racing a writer.
|
|
2265
|
+
if (this.pendingRestores.size > 0) {
|
|
2266
|
+
try {
|
|
2267
|
+
await withTimeout(
|
|
2268
|
+
Promise.allSettled([...this.pendingRestores]),
|
|
2269
|
+
Math.max(1, deadlineMs - systemClock()),
|
|
2270
|
+
`${this.pendingRestores.size} earlier restore(s) still running`,
|
|
2271
|
+
);
|
|
2272
|
+
} catch (err) {
|
|
2273
|
+
// Same exit as a stalled restore below: the stream is still running and
|
|
2274
|
+
// a rebuild is what follows a `down`, so the container goes with it.
|
|
2275
|
+
this.swapIncarnation(); // deliberate incarnation swap
|
|
2276
|
+
await this.stop().catch((stopErr) => console.log(`restore: stop failed: ${errMsg(stopErr)}`));
|
|
2277
|
+
throw await this.goDown(
|
|
2278
|
+
`r2-restore-failed: ${errMsg(err)} — container stopped so the transfer cannot land on a rebuild`,
|
|
2279
|
+
);
|
|
2280
|
+
}
|
|
2281
|
+
}
|
|
2282
|
+
// A previous incarnation's restore may still be MOUNTED at these paths
|
|
2283
|
+
// (item 61: the SDK's presigned restore mounts) — `rm -rf` on a mount
|
|
2284
|
+
// point is "Device or resource busy". Unmount first, every time.
|
|
2285
|
+
await this.runOk(["sh", "-c", unmountAllRestoresScript()], "unmount-restores");
|
|
2286
|
+
await this.runOk(["rm", "-rf", MIRROR_DIR, CHECKOUT_DIR, ...DISK_MARKERS], "clean-before-restore");
|
|
2287
|
+
try {
|
|
2288
|
+
// The restore pair IS the cold-wake critical path. Sequential on purpose:
|
|
2289
|
+
// the SDK serializes backup operations anyway (one queue), so a
|
|
2290
|
+
// concurrent pair only made the second one's clock run while it waited —
|
|
2291
|
+
// and each is judged by its own bytes (restoreWithProgress), not by
|
|
2292
|
+
// a fixed budget: a slow transfer waits, a stalled one goes down with
|
|
2293
|
+
// the bytes and the idle span named instead of stranding `restoring` for
|
|
2294
|
+
// the watchdog.
|
|
2295
|
+
await this.restoreExtracted(snap.mirror, MIRROR_DIR, "mirror restore", "mirror-restore-extract", deadlineMs);
|
|
2296
|
+
await this.restoreExtracted(
|
|
2297
|
+
snap.checkout,
|
|
2298
|
+
CHECKOUT_DIR,
|
|
2299
|
+
"checkout restore",
|
|
2300
|
+
"checkout-restore-extract",
|
|
2301
|
+
deadlineMs,
|
|
2302
|
+
);
|
|
2303
|
+
} catch (err) {
|
|
2304
|
+
// A stalled or capped restore is STILL STREAMING (the SDK call cannot be
|
|
2305
|
+
// cancelled); `pendingRestores` keeps the next hydrate off its directory,
|
|
2306
|
+
// but a `down` resident's only exit is a REBUILD, and provisioning owns
|
|
2307
|
+
// the same directories. Left running, the restore the wake path gave up
|
|
2308
|
+
// on lands into the checkout the rebuild has just cloned and linked —
|
|
2309
|
+
// tar overwrites in place through the deps store's hardlinks, resetting
|
|
2310
|
+
// every hardened entry file from 444 to 644. Stop
|
|
2311
|
+
// the container on the way down: the disk is ephemeral, the stream dies
|
|
2312
|
+
// with it, and the rebuild starts on an empty one.
|
|
2313
|
+
this.swapIncarnation(); // deliberate incarnation swap
|
|
2314
|
+
await this.stop().catch((stopErr) => console.log(`restore: stop failed: ${errMsg(stopErr)}`));
|
|
2315
|
+
throw await this.goDown(
|
|
2316
|
+
`r2-restore-failed: ${errMsg(err)} — container stopped so the transfer cannot land on a rebuild`,
|
|
2317
|
+
);
|
|
2318
|
+
}
|
|
2319
|
+
await this.ensureGitSetup();
|
|
2320
|
+
|
|
2321
|
+
// Verify the restored disk against the stamp — a snapshot that does not
|
|
2322
|
+
// prove its own {ref, sha, lockfileHash} is refused. Both values
|
|
2323
|
+
// derive from the restored MIRROR (the source of truth the checkout was
|
|
2324
|
+
// built from); the checkout's presence was proven by restoreBackup + the
|
|
2325
|
+
// chown below failing loudly if it is missing.
|
|
2326
|
+
const shaRes = await this.run(["git", "-C", MIRROR_DIR, "rev-parse", "--verify", `refs/heads/${snap.ref}`]);
|
|
2327
|
+
const diskSha = shaRes.exitCode === 0 ? shaRes.stdout.trim() : `unreadable(${tail(shaRes.stderr, 120)})`;
|
|
2328
|
+
const diskLock = await this.lockfileKey(snap.sha).catch((err) => `unreadable(${errMsg(err)})`);
|
|
2329
|
+
if (diskSha !== snap.sha || diskLock !== snap.lockfileHash) {
|
|
2330
|
+
throw await this.goDown(
|
|
2331
|
+
`snapshot-stamp-mismatch: restored disk {sha:${diskSha}, lockfileHash:${diskLock}} != stamp {sha:${snap.sha}, lockfileHash:${snap.lockfileHash}}`,
|
|
2332
|
+
);
|
|
2333
|
+
}
|
|
2334
|
+
|
|
2335
|
+
await this.runOk(["chown", "-R", `${BUILD_USER}:${BUILD_USER}`, CHECKOUT_DIR], "chown");
|
|
2336
|
+
return { done: false };
|
|
2337
|
+
}
|
|
2338
|
+
|
|
1782
2339
|
/** Delete the R2 objects behind SDK backup handles (backups/<id>/ lives
|
|
1783
2340
|
* OUTSIDE the resident/<resource>/ prefix, so offboard's prefix sweep
|
|
1784
2341
|
* cannot reach it — this is the only cleanup path). The ids' prefixes are
|
|
@@ -1922,17 +2479,22 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
1922
2479
|
// install budget is at least the refresh's: a provisioning budget below
|
|
1923
2480
|
// a real install time just fails the onboarding.
|
|
1924
2481
|
if (record.commands.install) {
|
|
1925
|
-
const entry = await this.
|
|
1926
|
-
lockfileHash,
|
|
2482
|
+
const { entry } = await this.installDeps({
|
|
2483
|
+
key: lockfileHash,
|
|
1927
2484
|
sha,
|
|
1928
|
-
record.commands.install,
|
|
1929
|
-
Math.max(stepBudget, REFRESH_INSTALL_TIMEOUT_MS),
|
|
1930
|
-
);
|
|
2485
|
+
installCmd: record.commands.install,
|
|
2486
|
+
budgetMs: Math.max(stepBudget, REFRESH_INSTALL_TIMEOUT_MS),
|
|
2487
|
+
});
|
|
1931
2488
|
await this.linkDepsView(`${entry}/node_modules`, CHECKOUT_DIR, BUILD_USER);
|
|
1932
2489
|
}
|
|
1933
2490
|
await this.buildUserRun(record.commands.build, "build", stepBudget);
|
|
1934
2491
|
|
|
1935
|
-
|
|
2492
|
+
// Provisioning is the only writer while `onboarding`, so the step cannot
|
|
2493
|
+
// be superseded; a record already at the stamp (a re-fired schedule) is
|
|
2494
|
+
// reused.
|
|
2495
|
+
const snapped = await this.snapshot({ resource, stamp: { ref, sha, lockfileHash } });
|
|
2496
|
+
if (!snapped.done && snapped.superseded) throw new StepError("snapshot", "superseded by a concurrent writer");
|
|
2497
|
+
const snap = snapped.record;
|
|
1936
2498
|
|
|
1937
2499
|
// The deadline may have fired mid-provision (down + slot released);
|
|
1938
2500
|
// never flip a non-onboarding resident to warm from here.
|
|
@@ -1997,12 +2559,18 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
1997
2559
|
// 10-min cadence always outlives the TTL, so a cycle re-probes for real.
|
|
1998
2560
|
if (this.hydratedVerdictAt !== 0 && systemClock() - this.hydratedVerdictAt < this.hydrationMemoTtlMs) return;
|
|
1999
2561
|
if (this.hydration) return this.hydration;
|
|
2000
|
-
|
|
2562
|
+
// The hydration's lease in the in-flight row (item 22) is what the
|
|
2563
|
+
// watchdog reads: alive for this incarnation until the stale bound, gone
|
|
2564
|
+
// with the isolate that started it.
|
|
2565
|
+
const holder = this.nextHolder();
|
|
2566
|
+
const p = this.recordInFlight("hydration", holder, STALE_MIDFLIGHT_MS, "restore")
|
|
2567
|
+
.then(() => this.doHydrate())
|
|
2001
2568
|
.then(() => {
|
|
2002
2569
|
this.hydratedVerdictAt = systemClock();
|
|
2003
2570
|
})
|
|
2004
|
-
.finally(() => {
|
|
2571
|
+
.finally(async () => {
|
|
2005
2572
|
if (this.hydration === p) this.hydration = null;
|
|
2573
|
+
await this.clearInFlight("hydration", holder);
|
|
2006
2574
|
});
|
|
2007
2575
|
this.hydration = p;
|
|
2008
2576
|
this.hydrationStartedAt = systemClock();
|
|
@@ -2044,93 +2612,17 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2044
2612
|
if (active && (await this.diskMatches(snap.sha))) return;
|
|
2045
2613
|
|
|
2046
2614
|
await this.setResidentState("restoring", "rehydrating");
|
|
2047
|
-
|
|
2615
|
+
const t0 = systemClock();
|
|
2616
|
+
// One deadline for the whole hydrate: the restore step's wait for earlier
|
|
2617
|
+
// restores, both restores and the deps materialization below judge
|
|
2618
|
+
// against it, so the worst-case `restoring` span is RESTORE_MAX_MS, under
|
|
2619
|
+
// the watchdog's stale-mid-flight window — not three caps in a row.
|
|
2620
|
+
const deadlineMs = systemClock() + RESTORE_MAX_MS;
|
|
2621
|
+
if ((await this.restoreCheckout(snap, deadlineMs)).done) {
|
|
2048
2622
|
// Raced a container start that already had the right disk.
|
|
2049
2623
|
await this.setResidentState("warm");
|
|
2050
2624
|
return;
|
|
2051
2625
|
}
|
|
2052
|
-
|
|
2053
|
-
const t0 = systemClock();
|
|
2054
|
-
// A restore a previous attempt gave up on may still be writing into these
|
|
2055
|
-
// directories (the SDK call cannot be cancelled): wait for it to
|
|
2056
|
-
// settle before the clean, bounded by the same cap the restores get. A
|
|
2057
|
-
// restore that will not settle even then leaves the disk alone — a named
|
|
2058
|
-
// `down`, not a clean racing a writer.
|
|
2059
|
-
// One deadline for the whole hydrate: the wait below and both restores
|
|
2060
|
-
// judge against it, so the worst-case `restoring` span is RESTORE_MAX_MS,
|
|
2061
|
-
// under the watchdog's stale-mid-flight window — not three caps in a row.
|
|
2062
|
-
const deadlineMs = systemClock() + RESTORE_MAX_MS;
|
|
2063
|
-
if (this.pendingRestores.size > 0) {
|
|
2064
|
-
try {
|
|
2065
|
-
await withTimeout(
|
|
2066
|
-
Promise.allSettled([...this.pendingRestores]),
|
|
2067
|
-
Math.max(1, deadlineMs - systemClock()),
|
|
2068
|
-
`${this.pendingRestores.size} earlier restore(s) still running`,
|
|
2069
|
-
);
|
|
2070
|
-
} catch (err) {
|
|
2071
|
-
// Same exit as a stalled restore below: the stream is still running and
|
|
2072
|
-
// a rebuild is what follows a `down`, so the container goes with it.
|
|
2073
|
-
this.clearIncarnationMemos(); // deliberate incarnation swap
|
|
2074
|
-
await this.stop().catch((stopErr) => console.log(`restore: stop failed: ${errMsg(stopErr)}`));
|
|
2075
|
-
throw await this.goDown(
|
|
2076
|
-
`r2-restore-failed: ${errMsg(err)} — container stopped so the transfer cannot land on a rebuild`,
|
|
2077
|
-
);
|
|
2078
|
-
}
|
|
2079
|
-
}
|
|
2080
|
-
// A previous incarnation's restore may still be MOUNTED at these paths
|
|
2081
|
-
// (item 61: the SDK's presigned restore mounts) — `rm -rf` on a mount
|
|
2082
|
-
// point is "Device or resource busy". Unmount first, every time.
|
|
2083
|
-
await this.runOk(["sh", "-c", unmountAllRestoresScript()], "unmount-restores");
|
|
2084
|
-
await this.runOk(["rm", "-rf", MIRROR_DIR, CHECKOUT_DIR, ...DISK_MARKERS], "clean-before-restore");
|
|
2085
|
-
try {
|
|
2086
|
-
// The restore pair IS the cold-wake critical path. Sequential on purpose:
|
|
2087
|
-
// the SDK serializes backup operations anyway (one queue), so a
|
|
2088
|
-
// concurrent pair only made the second one's clock run while it waited —
|
|
2089
|
-
// and each is judged by its own bytes (restoreWithProgress), not by
|
|
2090
|
-
// a fixed budget: a slow transfer waits, a stalled one goes down with
|
|
2091
|
-
// the bytes and the idle span named instead of stranding `restoring` for
|
|
2092
|
-
// the watchdog.
|
|
2093
|
-
await this.restoreExtracted(snap.mirror, MIRROR_DIR, "mirror restore", "mirror-restore-extract", deadlineMs);
|
|
2094
|
-
await this.restoreExtracted(
|
|
2095
|
-
snap.checkout,
|
|
2096
|
-
CHECKOUT_DIR,
|
|
2097
|
-
"checkout restore",
|
|
2098
|
-
"checkout-restore-extract",
|
|
2099
|
-
deadlineMs,
|
|
2100
|
-
);
|
|
2101
|
-
} catch (err) {
|
|
2102
|
-
// A stalled or capped restore is STILL STREAMING (the SDK call cannot be
|
|
2103
|
-
// cancelled); `pendingRestores` keeps the next hydrate off its directory,
|
|
2104
|
-
// but a `down` resident's only exit is a REBUILD, and provisioning owns
|
|
2105
|
-
// the same directories. Left running, the restore the wake path gave up
|
|
2106
|
-
// on lands into the checkout the rebuild has just cloned and linked —
|
|
2107
|
-
// tar overwrites in place through the deps store's hardlinks, resetting
|
|
2108
|
-
// every hardened entry file from 444 to 644. Stop
|
|
2109
|
-
// the container on the way down: the disk is ephemeral, the stream dies
|
|
2110
|
-
// with it, and the rebuild starts on an empty one.
|
|
2111
|
-
this.clearIncarnationMemos(); // deliberate incarnation swap
|
|
2112
|
-
await this.stop().catch((stopErr) => console.log(`restore: stop failed: ${errMsg(stopErr)}`));
|
|
2113
|
-
throw await this.goDown(
|
|
2114
|
-
`r2-restore-failed: ${errMsg(err)} — container stopped so the transfer cannot land on a rebuild`,
|
|
2115
|
-
);
|
|
2116
|
-
}
|
|
2117
|
-
await this.ensureGitSetup();
|
|
2118
|
-
|
|
2119
|
-
// Verify the restored disk against the stamp — a snapshot that does not
|
|
2120
|
-
// prove its own {ref, sha, lockfileHash} is refused. Both values
|
|
2121
|
-
// derive from the restored MIRROR (the source of truth the checkout was
|
|
2122
|
-
// built from); the checkout's presence was proven by restoreBackup + the
|
|
2123
|
-
// chown below failing loudly if it is missing.
|
|
2124
|
-
const shaRes = await this.run(["git", "-C", MIRROR_DIR, "rev-parse", "--verify", `refs/heads/${snap.ref}`]);
|
|
2125
|
-
const diskSha = shaRes.exitCode === 0 ? shaRes.stdout.trim() : `unreadable(${tail(shaRes.stderr, 120)})`;
|
|
2126
|
-
const diskLock = await this.lockfileKey(snap.sha).catch((err) => `unreadable(${errMsg(err)})`);
|
|
2127
|
-
if (diskSha !== snap.sha || diskLock !== snap.lockfileHash) {
|
|
2128
|
-
throw await this.goDown(
|
|
2129
|
-
`snapshot-stamp-mismatch: restored disk {sha:${diskSha}, lockfileHash:${diskLock}} != stamp {sha:${snap.sha}, lockfileHash:${snap.lockfileHash}}`,
|
|
2130
|
-
);
|
|
2131
|
-
}
|
|
2132
|
-
|
|
2133
|
-
await this.runOk(["chown", "-R", `${BUILD_USER}:${BUILD_USER}`, CHECKOUT_DIR], "chown");
|
|
2134
2626
|
// The snapshot carries the checkout's tree, not the store (item 59): adopt
|
|
2135
2627
|
// its node_modules as the entry for the stamp's key — a rename plus a
|
|
2136
2628
|
// hardlink view, seconds — so the first attach on the warm key hits.
|
|
@@ -2160,26 +2652,32 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2160
2652
|
const hasView = (await this.run(["test", "-d", `${CHECKOUT_DIR}/node_modules`])).exitCode === 0;
|
|
2161
2653
|
if (!record.commands.install) {
|
|
2162
2654
|
depsLinked = hasView;
|
|
2163
|
-
} else if (hasView) {
|
|
2164
|
-
depsLinked = true;
|
|
2165
2655
|
} else {
|
|
2166
|
-
const
|
|
2167
|
-
|
|
2168
|
-
|
|
2169
|
-
|
|
2656
|
+
const plan = planMaterializeDeps({
|
|
2657
|
+
key: snap.lockfileHash,
|
|
2658
|
+
viewPresent: hasView,
|
|
2659
|
+
entryComplete: await this.depsEntryComplete(snap.lockfileHash),
|
|
2170
2660
|
});
|
|
2171
|
-
if (
|
|
2172
|
-
|
|
2173
|
-
|
|
2174
|
-
|
|
2175
|
-
|
|
2176
|
-
|
|
2177
|
-
|
|
2661
|
+
if (plan.action === "done") {
|
|
2662
|
+
depsLinked = true;
|
|
2663
|
+
} else {
|
|
2664
|
+
console.log(`deps: ${plan.why}`);
|
|
2665
|
+
const budget = planWakeDepsBudget({
|
|
2666
|
+
nowMs: systemClock(),
|
|
2667
|
+
deadlineMs,
|
|
2668
|
+
installBudgetMs: REFRESH_INSTALL_TIMEOUT_MS,
|
|
2669
|
+
});
|
|
2670
|
+
if (budget.action === "skip") throw new Error(`${budget.remainingMs} ms left of the hydrate deadline`);
|
|
2671
|
+
const { entry } = await this.installDeps({
|
|
2672
|
+
key: snap.lockfileHash,
|
|
2673
|
+
sha: snap.sha,
|
|
2674
|
+
installCmd: record.commands.install,
|
|
2675
|
+
budgetMs: budget.installBudgetMs,
|
|
2178
2676
|
restoreDeadlineMs: deadlineMs,
|
|
2179
|
-
}
|
|
2180
|
-
|
|
2181
|
-
|
|
2182
|
-
|
|
2677
|
+
});
|
|
2678
|
+
await this.linkDepsView(`${entry}/node_modules`, CHECKOUT_DIR, BUILD_USER);
|
|
2679
|
+
depsLinked = true;
|
|
2680
|
+
}
|
|
2183
2681
|
}
|
|
2184
2682
|
} catch (err) {
|
|
2185
2683
|
console.log(`deps: no view after restore — the next refresh installs: ${errMsg(err)}`);
|
|
@@ -2228,331 +2726,88 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2228
2726
|
|
|
2229
2727
|
private async onRefreshAlarmTraced(payload: string): Promise<void> {
|
|
2230
2728
|
const resource = payload || ((await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "");
|
|
2729
|
+
// Item 7: a resident on the Workflow lifecycle has no chain. An alarm a
|
|
2730
|
+
// previous flip left armed — or an attach's +1 s pull, or a provisioning's
|
|
2731
|
+
// first arm — runs nothing and re-arms nothing, so the two schedulers never
|
|
2732
|
+
// both drive a cycle for one resident.
|
|
2733
|
+
if ((await this.getLifecycle()) === "workflow") {
|
|
2734
|
+
console.log(`refresh: lifecycle is workflow — the alarm chain runs no cycle for ${resource}`);
|
|
2735
|
+
return;
|
|
2736
|
+
}
|
|
2231
2737
|
let refreshCounted = false;
|
|
2738
|
+
/** This cycle's lease holder in the in-flight row, once it is counted. */
|
|
2739
|
+
let cycleHolder: string | null = null;
|
|
2740
|
+
/** This firing's identity: what `fetchMirror` records so a repeated call inside the cycle is done. */
|
|
2741
|
+
const cycle = crypto.randomUUID();
|
|
2232
2742
|
const before = await this.getStatus();
|
|
2233
2743
|
// down chains stay down (a rebuild is the escape hatch); onboarding is
|
|
2234
2744
|
// owned by provisioning, which arms the first refresh itself.
|
|
2235
2745
|
if (before.state === "onboarding" || before.state === "down") return;
|
|
2236
2746
|
try {
|
|
2237
|
-
await this.
|
|
2238
|
-
|
|
2239
|
-
|
|
2240
|
-
const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
|
|
2241
|
-
if (!facts) throw new StepError("facts", "no repo facts recorded despite hydration");
|
|
2242
|
-
|
|
2243
|
-
// Deploy-ordering hazard: `wrangler deploy` swaps the app's image but a
|
|
2244
|
-
// RUNNING container keeps the old one, so new Worker code can name pool
|
|
2245
|
-
// users the image lacks. Reconcile here (every cycle, cheap) — see
|
|
2246
|
-
// reconcileImage — so a rollout self-applies within one refresh.
|
|
2247
|
-
if (await this.reconcileImage("refresh")) {
|
|
2248
|
-
// Container stopping; it restarts on the new image in seconds. Re-arm
|
|
2249
|
-
// SHORT so the resident is re-warmed within a minute instead of
|
|
2250
|
-
// sitting on the old cadence for a full 600 s.
|
|
2251
|
-
this.rearmOutcome = "image-stale-restart";
|
|
2252
|
-
return; // finally re-arms
|
|
2253
|
-
}
|
|
2254
|
-
|
|
2255
|
-
// Idle sleep: nobody has attached for IDLE_AFTER_S and no live tree is
|
|
2256
|
-
// dirty → skip this fetch and park the alarm far out so SLEEP_AFTER can
|
|
2257
|
-
// elapse. Staleness is repaid at the next attach (refreshIfStale). A
|
|
2258
|
-
// dirty live tree pins the container awake: sleep destroys the disk and
|
|
2259
|
-
// uncommitted work is not snapshotted.
|
|
2260
|
-
// Only a SETTLED resident may park: a cycle that finds `refreshing`/
|
|
2261
|
-
// `restoring` at entry is looking at a marker left by a cycle that died
|
|
2262
|
-
// mid-flight (a deploy evicting the DO: stuck `refreshing` + parked →
|
|
2263
|
-
// every run falls back cold because the bot's warm-gate probe never
|
|
2264
|
-
// sees `warm` again). Run the full cycle instead; it
|
|
2265
|
-
// ends warm or degraded, and the next one may park.
|
|
2266
|
-
// Decide off a FRESH state read — `before` predates several awaits
|
|
2267
|
-
// (hydration, registry, facts, reconcile) — same re-read discipline as
|
|
2268
|
-
// every other state decision in this file.
|
|
2269
|
-
const entry = await this.getStatus();
|
|
2270
|
-
if (entry.state === "degraded" && isDiskFullReason(entry.reason)) {
|
|
2271
|
-
// The cycle owns the disk-full verdict (docs/reference/specs/resident-repos.md item 54): re-probe before
|
|
2272
|
-
// fetching. Still full → nothing a fetch can do; decide whether the
|
|
2273
|
-
// container may be recycled and stop here (a fetch that happened to fit
|
|
2274
|
-
// would flip the resident `warm`, the bot would attach, git-setup would
|
|
2275
|
-
// fail and flip it back — a flap loop). Space back (a detach or the
|
|
2276
|
-
// sweep freed trees) → run the cycle as usual and earn `warm`.
|
|
2277
|
-
const free = await this.freeKiB();
|
|
2278
|
-
if (free !== null && free < DISK_FULL_FREE_KIB) {
|
|
2279
|
-
await this.recoverFromDiskFull(entry.reason, 0);
|
|
2280
|
-
return; // finally re-arms: short after a recycle, the cadence otherwise
|
|
2281
|
-
}
|
|
2282
|
-
}
|
|
2283
|
-
let settled = entry.state === "warm";
|
|
2284
|
-
if (entry.state === "degraded" && !isNonEvidenceReason(entry.reason)) {
|
|
2285
|
-
// Count consecutive cycles that found the same REFRESH-PRODUCED degraded
|
|
2286
|
-
// reason (github-unreachable, <step>-failed); a stable streak means
|
|
2287
|
-
// retrying is not going to help and parking is the right cost behavior.
|
|
2288
|
-
// Any other state resets the streak (below).
|
|
2289
|
-
const prev = await this.ctx.storage.get<{ reason: string; count: number }>(DEGRADED_STREAK_KEY);
|
|
2290
|
-
const streak =
|
|
2291
|
-
prev && prev.reason === entry.reason
|
|
2292
|
-
? { reason: entry.reason, count: prev.count + 1 }
|
|
2293
|
-
: { reason: entry.reason, count: 1 };
|
|
2294
|
-
await this.ctx.storage.put(DEGRADED_STREAK_KEY, streak);
|
|
2295
|
-
settled = streak.count >= DEGRADED_PARK_AFTER_CYCLES;
|
|
2296
|
-
} else {
|
|
2297
|
-
// Warm, or a degraded stamped by the WATCHDOG (alarm-missed /
|
|
2298
|
-
// stale-mid-flight) or by an INTERRUPTED cycle (refresh-interrupted —
|
|
2299
|
-
// a deploy killed the step; it says nothing about the repo): the
|
|
2300
|
-
// watchdog pulled this cycle to +5s precisely so a refresh RUNS, and the
|
|
2301
|
-
// interrupted cycle re-armed short for the same reason.
|
|
2302
|
-
// Counting those toward the streak would be self-fulfilling —
|
|
2303
|
-
// each cycle that found the reason would park without attempting anything,
|
|
2304
|
-
// and after three the resident would sit parked-degraded for 6h at a
|
|
2305
|
-
// time. Never settled; streak reset.
|
|
2306
|
-
await this.ctx.storage.delete(DEGRADED_STREAK_KEY);
|
|
2307
|
-
}
|
|
2308
|
-
if (settled && (await this.isIdle())) {
|
|
2309
|
-
// isIdle awaited (git status per live tree) — re-read before writing.
|
|
2310
|
-
const now = (await this.ctx.storage.get<RepoFacts>(FACTS_KEY)) ?? facts;
|
|
2311
|
-
if (!now.idleSince)
|
|
2312
|
-
await this.ctx.storage.put(FACTS_KEY, {
|
|
2313
|
-
...now,
|
|
2314
|
-
idleSince: new Date(systemClock()).toISOString(),
|
|
2315
|
-
} satisfies RepoFacts);
|
|
2316
|
-
this.rearmOutcome = "idle";
|
|
2317
|
-
return; // finally re-arms at IDLE_REFRESH_INTERVAL_S
|
|
2318
|
-
}
|
|
2319
|
-
if (facts.idleSince) {
|
|
2320
|
-
const now = (await this.ctx.storage.get<RepoFacts>(FACTS_KEY)) ?? facts;
|
|
2321
|
-
const { idleSince: _woke, ...awake } = now;
|
|
2322
|
-
await this.ctx.storage.put(FACTS_KEY, awake satisfies RepoFacts);
|
|
2323
|
-
}
|
|
2747
|
+
const gate = await this.refreshGate(resource);
|
|
2748
|
+
if (!gate.go) return; // finally re-arms on the outcome the gate set
|
|
2749
|
+
const { record, facts } = gate;
|
|
2324
2750
|
// From here the cycle mutates the mirror/checkout: count it as in flight
|
|
2325
|
-
// so an attach-path reconcileImage never stops the container under it
|
|
2751
|
+
// so an attach-path reconcileImage never stops the container under it,
|
|
2752
|
+
// and lease it in the in-flight row so the watchdog can tell this cycle
|
|
2753
|
+
// from a marker a dead one left behind (item 22).
|
|
2326
2754
|
this.refreshesInFlight++;
|
|
2327
2755
|
refreshCounted = true;
|
|
2756
|
+
cycleHolder = this.nextHolder();
|
|
2757
|
+
await this.recordInFlight("refresh", cycleHolder, REFRESH_CYCLE_LEASE_MS, "refresh");
|
|
2328
2758
|
|
|
2329
|
-
|
|
2330
|
-
|
|
2331
|
-
|
|
2332
|
-
// fetch (exactly what an unconfigured App does): a public repo outside
|
|
2333
|
-
// the installation stays fresh, and a private one fails at the fetch
|
|
2334
|
-
// below into a visible `degraded(github-unreachable: …)`. Returning here
|
|
2335
|
-
// instead would freeze whatever state the resident was in — a public
|
|
2336
|
-
// repo the App is not installed on would sit in the watchdog's
|
|
2337
|
-
// `degraded(alarm-missed)` forever with an ever-staler mirror, because
|
|
2338
|
-
// the App cannot mint for a repo it is not installed on.
|
|
2339
|
-
let token: string | null = null;
|
|
2340
|
-
// This cycle's mint error, kept so it survives the warm facts write below
|
|
2341
|
-
// (which clears errors from PRIOR cycles) and prefixes a fetch failure's
|
|
2342
|
-
// reason — the observable for "App configured, repo outside the
|
|
2343
|
-
// installation" is a warm-but-anonymous resident with the mint named.
|
|
2344
|
-
let mintError: string | undefined;
|
|
2345
|
-
if (githubAppConfigured(this.env)) {
|
|
2346
|
-
try {
|
|
2347
|
-
token = (await mintRepoScopedToken(this.env, resource.slice("repo:".length))).token;
|
|
2348
|
-
} catch (err) {
|
|
2349
|
-
mintError = `token-mint-failed (command-level, fetching anonymously): ${errMsg(err)}`;
|
|
2350
|
-
await this.recordRefreshError(mintError);
|
|
2351
|
-
}
|
|
2352
|
-
}
|
|
2353
|
-
|
|
2354
|
-
await this.setResidentState("refreshing");
|
|
2355
|
-
try {
|
|
2356
|
-
// Same mirror mutex as attach's fetch/worktree work: the
|
|
2357
|
-
// refresh alarm and an in-flight attach serialize instead of racing
|
|
2358
|
-
// a prune against a worktree clone.
|
|
2359
|
-
await this.withMirrorLock(() =>
|
|
2360
|
-
this.gitWithCred(token, ["-C", MIRROR_DIR, "fetch", "--prune", "origin"], "fetch", GIT_NETWORK_TIMEOUT_MS),
|
|
2361
|
-
);
|
|
2362
|
-
} catch (err) {
|
|
2363
|
-
// A private repo whose mint failed lands here (the anonymous fetch is
|
|
2364
|
-
// refused): say so, rather than blaming GitHub reachability alone.
|
|
2365
|
-
const cause = mintError ? `${mintError}; then ` : "";
|
|
2366
|
-
const message = `${cause}${errMsg(err)}`;
|
|
2367
|
-
// A full disk fails this step too — the credential file is written
|
|
2368
|
-
// here (`ENOSPC` on /workspace/.resident/git-credentials would read as
|
|
2369
|
-
// github-unreachable, a SERVICEABLE reason, so every run would attach
|
|
2370
|
-
// and die at git-setup). Name the disk instead: not
|
|
2371
|
-
// serviceable, and the recovery below can free it.
|
|
2372
|
-
const failure = await this.classifyFailure("fetch", message);
|
|
2373
|
-
if (failure.diskFull) {
|
|
2374
|
-
await this.setResidentState("degraded", failure.reason);
|
|
2375
|
-
await this.recoverFromDiskFull(failure.reason, refreshCounted ? 1 : 0);
|
|
2376
|
-
return;
|
|
2377
|
-
}
|
|
2378
|
-
await this.setResidentState("degraded", `github-unreachable: ${message}`);
|
|
2379
|
-
return;
|
|
2380
|
-
}
|
|
2381
|
-
|
|
2382
|
-
const sha = await this.readMirrorSha(facts.defaultRef);
|
|
2383
|
-
// Pure function of the commit — computed from the mirror before
|
|
2384
|
-
// any checkout work so the planner can compare it to the deps marker.
|
|
2385
|
-
const lockfileHash = sha === facts.sha ? facts.lockfileHash : await this.lockfileKey(sha);
|
|
2759
|
+
const fetched = await this.refreshFetch(resource, facts, cycle, 1);
|
|
2760
|
+
if (!fetched.ok) return;
|
|
2761
|
+
const { sha, lockfileHash, mintError, token } = fetched;
|
|
2386
2762
|
const plan = planRefresh({
|
|
2387
2763
|
sha,
|
|
2388
2764
|
factsSha: facts.sha,
|
|
2389
2765
|
lockfileKey: lockfileHash,
|
|
2390
2766
|
disk: await this.readRefreshDisk(),
|
|
2391
2767
|
});
|
|
2392
|
-
|
|
2393
|
-
|
|
2768
|
+
// Whether this cycle's snapshot step committed (`superseded` means
|
|
2769
|
+
// another writer moved the record, whose facts then stand).
|
|
2770
|
+
let committed = true;
|
|
2394
2771
|
if (plan.action !== "unchanged") {
|
|
2395
2772
|
const t0 = systemClock();
|
|
2396
2773
|
console.log(`refresh: ${facts.sha.slice(0, 8)} → ${sha.slice(0, 8)}: ${plan.action} (${plan.why})`);
|
|
2397
|
-
// Serialize the CHECKOUT_DIR mutation on the mirror mutex (FIX 2):
|
|
2398
|
-
// materializeThreadDeps reads CHECKOUT_DIR via `cp -al` under the same
|
|
2399
|
-
// lock, so an attach/op dep-copy can no longer hardlink a half-rebuilt
|
|
2400
|
-
// checkout into a thread tree (torn cache → false ❌ from `repo test`).
|
|
2401
|
-
// No wait timeout, exactly like the fetch lock above: the background
|
|
2402
|
-
// refresh queues behind an in-flight attach instead of flipping to
|
|
2403
|
-
// degraded on transient lock contention.
|
|
2404
|
-
// Token-free from here on: repo code runs during install/build.
|
|
2405
|
-
//
|
|
2406
2774
|
// Deps come from the store (item 59): a changed lockfile key is
|
|
2407
2775
|
// materialized ONCE into `/workspace/deps/<key>` — OUTSIDE the mirror
|
|
2408
2776
|
// lock, because the install runs in its own scratch clone and touches
|
|
2409
2777
|
// no consumer's tree (the staging step) — and the checkout's
|
|
2410
|
-
// node_modules becomes a hardlink view of that entry. An
|
|
2411
|
-
// needs the same key joins this very install instead of
|
|
2412
|
-
// own. Checkpoint: the deps marker comes off BEFORE the
|
|
2413
|
-
// interruption mid-install can never read as completion.
|
|
2778
|
+
// node_modules becomes a hardlink view of that entry (runBuild). An
|
|
2779
|
+
// attach that needs the same key joins this very install instead of
|
|
2780
|
+
// starting its own. Checkpoint: the deps marker comes off BEFORE the
|
|
2781
|
+
// install so an interruption mid-install can never read as completion.
|
|
2414
2782
|
let depsEntry: string | null = null;
|
|
2415
2783
|
if (plan.action === "rebuild") {
|
|
2416
|
-
await this.
|
|
2417
|
-
if (plan.install
|
|
2418
|
-
// The installing marker brackets the install (item 57): written
|
|
2419
|
-
// before, removed after the deps key lands, so a cycle that ends in
|
|
2420
|
-
// between is planned as a resume. The install is seeded from the
|
|
2421
|
-
// key the checkout holds now — npm reconciles the delta; a resumed
|
|
2422
|
-
// install finds its key already in the store when the last attempt
|
|
2423
|
-
// completed, or reconciles from the warm key again when it did not.
|
|
2424
|
-
await this.writeDiskMarkers({ installingKey: lockfileHash });
|
|
2425
|
-
depsEntry = await this.materializeDeps(
|
|
2426
|
-
lockfileHash,
|
|
2427
|
-
sha,
|
|
2428
|
-
record.commands.install,
|
|
2429
|
-
REFRESH_INSTALL_TIMEOUT_MS,
|
|
2430
|
-
{
|
|
2431
|
-
seedFromKey: facts.lockfileHash,
|
|
2432
|
-
},
|
|
2433
|
-
);
|
|
2434
|
-
}
|
|
2784
|
+
await this.refreshClearMarkers(plan.install);
|
|
2785
|
+
if (plan.install) depsEntry = await this.refreshInstall(record, facts, sha, lockfileHash);
|
|
2435
2786
|
}
|
|
2436
|
-
|
|
2437
|
-
|
|
2438
|
-
|
|
2439
|
-
|
|
2440
|
-
|
|
2441
|
-
|
|
2442
|
-
|
|
2443
|
-
|
|
2444
|
-
|
|
2445
|
-
// owner-read-only (deps-harden), so a write through them fails
|
|
2446
|
-
// loudly instead of silently reaching the store; the tool caches
|
|
2447
|
-
// inside node_modules are the checkout's private copies (item 18).
|
|
2448
|
-
//
|
|
2449
|
-
// Install gate: when the committed lockfile key is unchanged,
|
|
2450
|
-
// node_modules (the view) is excluded from the clean and no deps
|
|
2451
|
-
// work happens; a changed key takes the full clean and re-links the
|
|
2452
|
-
// view to the new entry — which is also what drops deps the new
|
|
2453
|
-
// lockfile no longer has.
|
|
2454
|
-
await this.buildUserRun(checkoutUpdateCommand(sha, plan.clean), "checkout-update", GIT_NETWORK_TIMEOUT_MS);
|
|
2455
|
-
if (plan.install) {
|
|
2456
|
-
// The old view (a resumed install's keep-deps clean leaves it in
|
|
2457
|
-
// place, item 57) makes way for the new entry's: hardlinks only,
|
|
2458
|
-
// the entry's inodes are untouched.
|
|
2459
|
-
if (depsEntry) {
|
|
2460
|
-
await this.runOk(["rm", "-rf", `${CHECKOUT_DIR}/node_modules`], "unlink-deps-view");
|
|
2461
|
-
await this.linkDepsView(`${depsEntry}/node_modules`, CHECKOUT_DIR, BUILD_USER);
|
|
2462
|
-
}
|
|
2463
|
-
await this.writeDiskMarkers({ depsKey: lockfileHash });
|
|
2464
|
-
await this.runOk(["rm", "-f", INSTALLING_MARKER], "clear-installing-marker");
|
|
2465
|
-
}
|
|
2466
|
-
await this.buildUserRun(record.commands.build, "build", REFRESH_BUILD_TIMEOUT_MS);
|
|
2467
|
-
await this.writeDiskMarkers({ builtSha: sha });
|
|
2468
|
-
}
|
|
2469
|
-
// `reuse`: the checkout already holds this sha with its deps and
|
|
2470
|
-
// build (an interrupted cycle got that far) — only the snapshot,
|
|
2471
|
-
// facts and stamp are missing, and they must still move together.
|
|
2472
|
-
previous = await this.ctx.storage.get<SnapshotRecord>(SNAPSHOT_KEY);
|
|
2473
|
-
snap = await this.takeSnapshot(resource, facts.defaultRef, sha, lockfileHash);
|
|
2787
|
+
// `reuse`: the checkout already holds this sha with its deps and build
|
|
2788
|
+
// (an interrupted cycle got that far) — the build step finds it done and
|
|
2789
|
+
// only the snapshot, facts and stamp are missing; they move together.
|
|
2790
|
+
await this.runBuild({
|
|
2791
|
+
sha,
|
|
2792
|
+
factsSha: facts.sha,
|
|
2793
|
+
lockfileKey: lockfileHash,
|
|
2794
|
+
buildCmd: record.commands.build,
|
|
2795
|
+
depsEntry,
|
|
2474
2796
|
});
|
|
2797
|
+
committed = await this.refreshSnapshot(resource, { ref: facts.defaultRef, sha, lockfileHash });
|
|
2475
2798
|
console.log(`refresh: ${sha.slice(0, 8)} ${plan.action} done in ${systemClock() - t0}ms`);
|
|
2476
2799
|
}
|
|
2477
|
-
|
|
2478
|
-
// Facts and snapshot move together so the stamp check never sees a
|
|
2479
|
-
// half-updated pair.
|
|
2480
|
-
const updatedFacts: RepoFacts = {
|
|
2481
|
-
...facts,
|
|
2482
|
-
sha,
|
|
2483
|
-
lockfileHash,
|
|
2484
|
-
lastRefreshAt: new Date(systemClock()).toISOString(),
|
|
2485
|
-
};
|
|
2486
|
-
// Clear a PRIOR cycle's error; keep THIS cycle's mint error visible.
|
|
2487
|
-
delete updatedFacts.lastRefreshError;
|
|
2488
|
-
if (mintError) updatedFacts.lastRefreshError = mintError;
|
|
2489
|
-
// A wake cycle cleared idleSince above; `facts` was read at alarm entry and
|
|
2490
|
-
// still carries it — never resurrect it here (the dash would show a stale
|
|
2491
|
-
// "idle since" and every attach would take the wake-fetch path).
|
|
2492
|
-
delete updatedFacts.idleSince;
|
|
2493
|
-
if (snap) {
|
|
2494
|
-
await this.ctx.storage.put({ [FACTS_KEY]: updatedFacts, [SNAPSHOT_KEY]: snap });
|
|
2495
|
-
await this.writeDiskMarkers({ ready: sha });
|
|
2496
|
-
if (previous) await this.deleteBackupObjects([previous.mirror.id, previous.checkout.id]).catch(() => {});
|
|
2497
|
-
} else {
|
|
2498
|
-
await this.ctx.storage.put(FACTS_KEY, updatedFacts);
|
|
2499
|
-
}
|
|
2500
|
-
await this.setResidentState("warm");
|
|
2501
|
-
// Event-triggered reclamation: the prune above already told the
|
|
2502
|
-
// mirror which branches died; finished refs give their worktree and
|
|
2503
|
-
// pool user back now, not at the idle TTL. Housekeeping, never a
|
|
2504
|
-
// lifecycle flip — a failure here is a log line.
|
|
2505
|
-
try {
|
|
2506
|
-
const gc = await this.reclaimFinishedRefs(resource, facts.defaultRef, token);
|
|
2507
|
-
if (gc.reclaimed.length > 0) console.log(`reclaim ${resource}: ${JSON.stringify(gc)}`);
|
|
2508
|
-
} catch (err) {
|
|
2509
|
-
console.log(`reclaim ${resource}: pass failed: ${errMsg(err)}`);
|
|
2510
|
-
}
|
|
2511
|
-
// Item 55: the cycle's disk sample — what /residents, `repo list`, the
|
|
2512
|
-
// watchdog line and the next attach admission read. Housekeeping too.
|
|
2513
|
-
await this.measureDisk().catch((err) => console.log(`disk: measure failed: ${errMsg(err)}`));
|
|
2800
|
+
await this.refreshComplete(resource, facts, { sha, lockfileHash, committed, mintError, token });
|
|
2514
2801
|
} catch (err) {
|
|
2515
2802
|
if (err instanceof ResidentDownError) return; // already down with reason; chain stops below
|
|
2516
|
-
|
|
2517
|
-
// image-changing deploy or a container stop; a Worker-only deploy leaves
|
|
2518
|
-
// the container running and interrupts nothing) is
|
|
2519
|
-
// `refresh-interrupted`: it is not evidence about the repo — it
|
|
2520
|
-
// never counts toward the park streak (the entry gate above) — and the
|
|
2521
|
-
// chain re-arms SHORT so the resident is warm again within a minute
|
|
2522
|
-
// instead of after the full cadence (an unclassified kill otherwise
|
|
2523
|
-
// costs the resident the whole 10-minute cadence, e.g.
|
|
2524
|
-
// `degraded(build-failed: exit 143 …)` until the next alarm).
|
|
2525
|
-
// Any other failure is the repo's own: `<step>-failed: …` /
|
|
2526
|
-
// `refresh-failed: …` as before. Non-StepErrors classify too — an SDK
|
|
2527
|
-
// replacement error can surface between steps — with the generic
|
|
2528
|
-
// "refresh" step, whose failure reason is the pre-existing
|
|
2529
|
-
// `refresh-failed: …` shape.
|
|
2530
|
-
// A full disk is a third class: `disk-full: …`, never serviceable,
|
|
2531
|
-
// and the one failure the resident can act on itself (recoverFromDiskFull).
|
|
2532
|
-
const failure =
|
|
2533
|
-
err instanceof StepError
|
|
2534
|
-
? await this.classifyFailure(err.step, err.message)
|
|
2535
|
-
: await this.classifyFailure("refresh", errMsg(err));
|
|
2803
|
+
const failure = await this.classifyCycleError(err);
|
|
2536
2804
|
// Set BEFORE the writes on purpose: if either throws, the finally still
|
|
2537
2805
|
// re-arms short — the safe direction for an interruption.
|
|
2538
2806
|
if (failure.interrupted) this.rearmOutcome = "interrupted";
|
|
2539
|
-
|
|
2540
|
-
// output block above, but a failure between steps (an SDK error, the
|
|
2541
|
-
// markers, the snapshot) reached only the state entry — which the next
|
|
2542
|
-
// cycle's failure overwrites (an install timeout that starts an
|
|
2543
|
-
// incident leaves no trace once the follow-up cycle fails).
|
|
2544
|
-
console.log(`refresh: cycle failed — ${failure.reason.slice(0, 400)}`);
|
|
2545
|
-
// Record on the facts too: the degraded state write below can be
|
|
2546
|
-
// clobbered within seconds by a concurrent attach/exec whose
|
|
2547
|
-
// ensureHydrated flips the state to `restoring · rehydrating`, leaving
|
|
2548
|
-
// no visible trace of WHY.
|
|
2549
|
-
// `lastRefreshError` survives that race and the next completed cycle
|
|
2550
|
-
// clears it, same as a mint error.
|
|
2551
|
-
await this.recordRefreshError(failure.reason);
|
|
2552
|
-
await this.setResidentState("degraded", failure.reason); // last snapshot keeps serving
|
|
2553
|
-
if (failure.diskFull) await this.recoverFromDiskFull(failure.reason, refreshCounted ? 1 : 0);
|
|
2807
|
+
await this.refreshFailed(failure, refreshCounted ? 1 : 0);
|
|
2554
2808
|
} finally {
|
|
2555
2809
|
if (refreshCounted) this.refreshesInFlight--;
|
|
2810
|
+
if (cycleHolder) await this.clearInFlight("refresh", cycleHolder);
|
|
2556
2811
|
const state = await this.ctx.storage.get<ResidentState>(STATE_KEY);
|
|
2557
2812
|
// Consecutive-interruption count: bounds the short re-arm so
|
|
2558
2813
|
// a step whose output chronically carries the kill signature falls back to
|
|
@@ -2571,10 +2826,685 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2571
2826
|
consecutiveInterrupted,
|
|
2572
2827
|
});
|
|
2573
2828
|
this.rearmOutcome = "normal";
|
|
2574
|
-
|
|
2829
|
+
// A flip to `workflow` while this cycle ran: the chain ends here (item 7).
|
|
2830
|
+
const chained = (await this.getLifecycle()) === "alarm";
|
|
2831
|
+
if (chained && state && state !== "down" && state !== "onboarding") await this.armRefresh(resource, interval);
|
|
2832
|
+
}
|
|
2833
|
+
}
|
|
2834
|
+
|
|
2835
|
+
// -- the cycle's phases, shared by the alarm and the instance (item 7) -------
|
|
2836
|
+
//
|
|
2837
|
+
// The alarm chain and the Workflow instance run the same cycle in the same
|
|
2838
|
+
// order: the gates, the fetch, the plan, the install, the build, the
|
|
2839
|
+
// snapshot, the completion. Each phase is one method here so neither
|
|
2840
|
+
// scheduler carries a copy; the instance calls them one step at a time
|
|
2841
|
+
// (`refreshInstance*`, below), the alarm in one handler (above).
|
|
2842
|
+
|
|
2843
|
+
/** The cycle's entry gates, in the alarm's order: hydrate; the registry
|
|
2844
|
+
* record (gone → the resident was offboarded mid-flight, nothing to do);
|
|
2845
|
+
* the image reconcile (a stale image stops the container, which restarts
|
|
2846
|
+
* on the current one — `image-stale-restart`); the disk-full re-probe (a
|
|
2847
|
+
* disk still full decides its own recovery and stops the cycle — item 54);
|
|
2848
|
+
* the park streak and the idle gate (`idle`); and, for a cycle that runs,
|
|
2849
|
+
* the end of idle mode. Sets `rearmOutcome` for the alarm's finally; the
|
|
2850
|
+
* instance resets it. */
|
|
2851
|
+
private async refreshGate(
|
|
2852
|
+
resource: string,
|
|
2853
|
+
): Promise<{ go: false; why: string } | { go: true; record: ResidentRecord; facts: RepoFacts }> {
|
|
2854
|
+
await this.ensureHydrated();
|
|
2855
|
+
const record = await this.registry().getRecord(resource);
|
|
2856
|
+
if (!record) return { go: false, why: "offboarded" }; // offboarded mid-flight: let the chain die quietly
|
|
2857
|
+
const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
|
|
2858
|
+
if (!facts) throw new StepError("facts", "no repo facts recorded despite hydration");
|
|
2859
|
+
|
|
2860
|
+
// Deploy-ordering hazard: `wrangler deploy` swaps the app's image but a
|
|
2861
|
+
// RUNNING container keeps the old one, so new Worker code can name pool
|
|
2862
|
+
// users the image lacks. Reconcile here (every cycle, cheap) — see
|
|
2863
|
+
// reconcileImage — so a rollout self-applies within one refresh.
|
|
2864
|
+
if (await this.reconcileImage("refresh")) {
|
|
2865
|
+
// Container stopping; it restarts on the new image in seconds. Re-arm
|
|
2866
|
+
// SHORT so the resident is re-warmed within a minute instead of
|
|
2867
|
+
// sitting on the old cadence for a full 600 s.
|
|
2868
|
+
this.rearmOutcome = "image-stale-restart";
|
|
2869
|
+
return { go: false, why: "image-stale-restart" };
|
|
2870
|
+
}
|
|
2871
|
+
|
|
2872
|
+
// Idle sleep: nobody has attached for IDLE_AFTER_S and no live tree is
|
|
2873
|
+
// dirty → skip this fetch and park the alarm far out so SLEEP_AFTER can
|
|
2874
|
+
// elapse. Staleness is repaid at the next attach (refreshIfStale). A
|
|
2875
|
+
// dirty live tree pins the container awake: sleep destroys the disk and
|
|
2876
|
+
// uncommitted work is not snapshotted.
|
|
2877
|
+
// Only a SETTLED resident may park: a cycle that finds `refreshing`/
|
|
2878
|
+
// `restoring` at entry is looking at a marker left by a cycle that died
|
|
2879
|
+
// mid-flight (a deploy evicting the DO: stuck `refreshing` + parked →
|
|
2880
|
+
// every run falls back cold because the bot's warm-gate probe never
|
|
2881
|
+
// sees `warm` again). Run the full cycle instead; it
|
|
2882
|
+
// ends warm or degraded, and the next one may park.
|
|
2883
|
+
// Decide off a FRESH state read — the caller's read predates several awaits
|
|
2884
|
+
// (hydration, registry, facts, reconcile) — same re-read discipline as
|
|
2885
|
+
// every other state decision in this file.
|
|
2886
|
+
const entry = await this.getStatus();
|
|
2887
|
+
if (entry.state === "degraded" && isDiskFullReason(entry.reason)) {
|
|
2888
|
+
// The cycle owns the disk-full verdict (docs/reference/specs/resident-repos.md item 54): re-probe before
|
|
2889
|
+
// fetching. Still full → nothing a fetch can do; decide whether the
|
|
2890
|
+
// container may be recycled and stop here (a fetch that happened to fit
|
|
2891
|
+
// would flip the resident `warm`, the bot would attach, git-setup would
|
|
2892
|
+
// fail and flip it back — a flap loop). Space back (a detach or the
|
|
2893
|
+
// sweep freed trees) → run the cycle as usual and earn `warm`.
|
|
2894
|
+
const free = await this.freeKiB();
|
|
2895
|
+
if (free !== null && free < DISK_FULL_FREE_KIB) {
|
|
2896
|
+
await this.recoverFromDiskFull(entry.reason, 0);
|
|
2897
|
+
return { go: false, why: "disk-full" }; // the alarm's finally re-arms: short after a recycle, the cadence otherwise
|
|
2898
|
+
}
|
|
2899
|
+
}
|
|
2900
|
+
let settled = entry.state === "warm";
|
|
2901
|
+
if (entry.state === "degraded" && !isNonEvidenceReason(entry.reason)) {
|
|
2902
|
+
// Count consecutive cycles that found the same REFRESH-PRODUCED degraded
|
|
2903
|
+
// reason (github-unreachable, <step>-failed); a stable streak means
|
|
2904
|
+
// retrying is not going to help and parking is the right cost behavior.
|
|
2905
|
+
// Any other state resets the streak (below).
|
|
2906
|
+
const prev = await this.ctx.storage.get<{ reason: string; count: number }>(DEGRADED_STREAK_KEY);
|
|
2907
|
+
const streak =
|
|
2908
|
+
prev && prev.reason === entry.reason
|
|
2909
|
+
? { reason: entry.reason, count: prev.count + 1 }
|
|
2910
|
+
: { reason: entry.reason, count: 1 };
|
|
2911
|
+
await this.ctx.storage.put(DEGRADED_STREAK_KEY, streak);
|
|
2912
|
+
settled = streak.count >= DEGRADED_PARK_AFTER_CYCLES;
|
|
2913
|
+
} else {
|
|
2914
|
+
// Warm, or a degraded stamped by the WATCHDOG (alarm-missed /
|
|
2915
|
+
// stale-mid-flight) or by an INTERRUPTED cycle (refresh-interrupted —
|
|
2916
|
+
// a deploy killed the step; it says nothing about the repo): the
|
|
2917
|
+
// watchdog pulled this cycle to +5s precisely so a refresh RUNS, and the
|
|
2918
|
+
// interrupted cycle re-armed short for the same reason.
|
|
2919
|
+
// Counting those toward the streak would be self-fulfilling —
|
|
2920
|
+
// each cycle that found the reason would park without attempting anything,
|
|
2921
|
+
// and after three the resident would sit parked-degraded for 6h at a
|
|
2922
|
+
// time. Never settled; streak reset.
|
|
2923
|
+
await this.ctx.storage.delete(DEGRADED_STREAK_KEY);
|
|
2924
|
+
}
|
|
2925
|
+
if (settled && (await this.isIdle())) {
|
|
2926
|
+
// isIdle awaited (git status per live tree) — re-read before writing.
|
|
2927
|
+
const now = (await this.ctx.storage.get<RepoFacts>(FACTS_KEY)) ?? facts;
|
|
2928
|
+
if (!now.idleSince)
|
|
2929
|
+
await this.ctx.storage.put(FACTS_KEY, {
|
|
2930
|
+
...now,
|
|
2931
|
+
idleSince: new Date(systemClock()).toISOString(),
|
|
2932
|
+
} satisfies RepoFacts);
|
|
2933
|
+
this.rearmOutcome = "idle";
|
|
2934
|
+
return { go: false, why: "idle" }; // the alarm's finally re-arms at IDLE_REFRESH_INTERVAL_S
|
|
2935
|
+
}
|
|
2936
|
+
if (facts.idleSince) {
|
|
2937
|
+
const now = (await this.ctx.storage.get<RepoFacts>(FACTS_KEY)) ?? facts;
|
|
2938
|
+
const { idleSince: _woke, ...awake } = now;
|
|
2939
|
+
await this.ctx.storage.put(FACTS_KEY, awake satisfies RepoFacts);
|
|
2940
|
+
}
|
|
2941
|
+
return { go: true, record, facts };
|
|
2942
|
+
}
|
|
2943
|
+
|
|
2944
|
+
/** The cycle's fetch phase: mint, `refreshing`, fetch, the lockfile key at
|
|
2945
|
+
* the new tip. Token-mint failure is a command-level error — the resident
|
|
2946
|
+
* keeps serving the last snapshot and lifecycle state is NOT flipped by
|
|
2947
|
+
* it. It is recorded, and the cycle then CONTINUES with an anonymous
|
|
2948
|
+
* fetch (exactly what an unconfigured App does): a public repo outside
|
|
2949
|
+
* the installation stays fresh, and a private one fails at the fetch
|
|
2950
|
+
* below into a visible `degraded(github-unreachable: …)`. Returning early
|
|
2951
|
+
* instead would freeze whatever state the resident was in — a public
|
|
2952
|
+
* repo the App is not installed on would sit in the watchdog's
|
|
2953
|
+
* `degraded(alarm-missed)` forever with an ever-staler mirror, because
|
|
2954
|
+
* the App cannot mint for a repo it is not installed on. A failed fetch
|
|
2955
|
+
* is recorded here — `degraded` with its reason, the disk-full recovery
|
|
2956
|
+
* when that is the cause — and answered `ok: false`. */
|
|
2957
|
+
private async refreshFetch(
|
|
2958
|
+
resource: string,
|
|
2959
|
+
facts: RepoFacts,
|
|
2960
|
+
cycle: string,
|
|
2961
|
+
selfInFlight: number,
|
|
2962
|
+
): Promise<
|
|
2963
|
+
| { ok: false; reason: string }
|
|
2964
|
+
| { ok: true; sha: string; lockfileHash: string; mintError: string | undefined; token: string | null }
|
|
2965
|
+
> {
|
|
2966
|
+
let token: string | null = null;
|
|
2967
|
+
// This cycle's mint error, kept so it survives the warm facts write
|
|
2968
|
+
// (which clears errors from PRIOR cycles) and prefixes a fetch failure's
|
|
2969
|
+
// reason — the observable for "App configured, repo outside the
|
|
2970
|
+
// installation" is a warm-but-anonymous resident with the mint named.
|
|
2971
|
+
let mintError: string | undefined;
|
|
2972
|
+
if (githubAppConfigured(this.env)) {
|
|
2973
|
+
try {
|
|
2974
|
+
token = (await mintRepoScopedToken(this.env, resource.slice("repo:".length))).token;
|
|
2975
|
+
} catch (err) {
|
|
2976
|
+
mintError = `token-mint-failed (command-level, fetching anonymously): ${errMsg(err)}`;
|
|
2977
|
+
await this.recordRefreshError(mintError);
|
|
2978
|
+
}
|
|
2979
|
+
}
|
|
2980
|
+
|
|
2981
|
+
await this.setResidentState("refreshing");
|
|
2982
|
+
let sha: string;
|
|
2983
|
+
try {
|
|
2984
|
+
// Same mirror mutex as attach's fetch/worktree work: the
|
|
2985
|
+
// refresh alarm and an in-flight attach serialize instead of racing
|
|
2986
|
+
// a prune against a worktree clone.
|
|
2987
|
+
sha = (await this.fetchMirror({ ref: facts.defaultRef, cycle, token })).sha;
|
|
2988
|
+
} catch (err) {
|
|
2989
|
+
// The fetch itself failed — or the tip could not be read afterwards,
|
|
2990
|
+
// which is the mirror's own failure, not GitHub's: the cycle's
|
|
2991
|
+
// classifier names that step.
|
|
2992
|
+
if (err instanceof StepError && err.step === "rev-parse") throw err;
|
|
2993
|
+
// A private repo whose mint failed lands here (the anonymous fetch is
|
|
2994
|
+
// refused): say so, rather than blaming GitHub reachability alone.
|
|
2995
|
+
const cause = mintError ? `${mintError}; then ` : "";
|
|
2996
|
+
const message = `${cause}${errMsg(err)}`;
|
|
2997
|
+
// A full disk fails this step too — the credential file is written
|
|
2998
|
+
// here (`ENOSPC` on /workspace/.resident/git-credentials would read as
|
|
2999
|
+
// github-unreachable, a SERVICEABLE reason, so every run would attach
|
|
3000
|
+
// and die at git-setup). Name the disk instead: not
|
|
3001
|
+
// serviceable, and the recovery below can free it.
|
|
3002
|
+
const failure = await this.classifyFailure("fetch", message);
|
|
3003
|
+
if (failure.diskFull) {
|
|
3004
|
+
await this.setResidentState("degraded", failure.reason);
|
|
3005
|
+
await this.recoverFromDiskFull(failure.reason, selfInFlight);
|
|
3006
|
+
return { ok: false, reason: failure.reason };
|
|
3007
|
+
}
|
|
3008
|
+
const reason = `github-unreachable: ${message}`;
|
|
3009
|
+
await this.setResidentState("degraded", reason);
|
|
3010
|
+
return { ok: false, reason };
|
|
3011
|
+
}
|
|
3012
|
+
|
|
3013
|
+
// Pure function of the commit — computed from the mirror before
|
|
3014
|
+
// any checkout work so the planner can compare it to the deps marker.
|
|
3015
|
+
const lockfileHash = sha === facts.sha ? facts.lockfileHash : await this.lockfileKey(sha);
|
|
3016
|
+
return { ok: true, sha, lockfileHash, mintError, token };
|
|
3017
|
+
}
|
|
3018
|
+
|
|
3019
|
+
/** A rebuild's first command: the markers for the steps about to be redone
|
|
3020
|
+
* come off — `built` always, `deps-key` only when the install runs — so
|
|
3021
|
+
* an interruption mid-step can never read as completion (item 48). */
|
|
3022
|
+
private async refreshClearMarkers(install: boolean): Promise<void> {
|
|
3023
|
+
await this.runOk(["rm", "-f", BUILT_MARKER, ...(install ? [DEPS_MARKER] : [])], "clear-markers");
|
|
3024
|
+
}
|
|
3025
|
+
|
|
3026
|
+
/** The cycle's install phase: the store entry for the new lockfile key,
|
|
3027
|
+
* bracketed by the installing marker (item 57) — written before, removed
|
|
3028
|
+
* by the build once the deps key lands, so a cycle that ends in between is
|
|
3029
|
+
* planned as a resume. The install is seeded from the key the checkout
|
|
3030
|
+
* holds now — npm reconciles the delta; a resumed install finds its key
|
|
3031
|
+
* already in the store when the last attempt completed, or reconciles
|
|
3032
|
+
* from the warm key again when it did not. Null when the command table
|
|
3033
|
+
* has no install: the build then links nothing. */
|
|
3034
|
+
private async refreshInstall(
|
|
3035
|
+
record: ResidentRecord,
|
|
3036
|
+
facts: RepoFacts,
|
|
3037
|
+
sha: string,
|
|
3038
|
+
lockfileHash: string,
|
|
3039
|
+
): Promise<string | null> {
|
|
3040
|
+
if (!record.commands.install) return null;
|
|
3041
|
+
await this.writeDiskMarkers({ installingKey: lockfileHash });
|
|
3042
|
+
const { entry } = await this.installDeps({
|
|
3043
|
+
key: lockfileHash,
|
|
3044
|
+
sha,
|
|
3045
|
+
installCmd: record.commands.install,
|
|
3046
|
+
budgetMs: REFRESH_INSTALL_TIMEOUT_MS,
|
|
3047
|
+
seedFromKey: facts.lockfileHash,
|
|
3048
|
+
});
|
|
3049
|
+
return entry;
|
|
3050
|
+
}
|
|
3051
|
+
|
|
3052
|
+
/** The cycle's snapshot phase: the stamped pair to R2 and, when this cycle's
|
|
3053
|
+
* record stands (not superseded by another writer's), the ready marker and
|
|
3054
|
+
* the replaced snapshot's objects swept. Answers whether the record committed. */
|
|
3055
|
+
private async refreshSnapshot(resource: string, stamp: SnapshotStamp): Promise<boolean> {
|
|
3056
|
+
const snapped = await this.snapshot({ resource, stamp });
|
|
3057
|
+
const committed = snapped.done || !snapped.superseded;
|
|
3058
|
+
if (committed) {
|
|
3059
|
+
await this.writeDiskMarkers({ ready: stamp.sha });
|
|
3060
|
+
const previous = !snapped.done && !snapped.superseded ? snapped.previous : undefined;
|
|
3061
|
+
if (previous) await this.deleteBackupObjects([previous.mirror.id, previous.checkout.id]).catch(() => {});
|
|
3062
|
+
}
|
|
3063
|
+
return committed;
|
|
3064
|
+
}
|
|
3065
|
+
|
|
3066
|
+
/** The cycle's completion: the facts to the stamp (the snapshot step moved
|
|
3067
|
+
* them in the same write as the record — a wake never sees a half-updated
|
|
3068
|
+
* pair; this is the cycle's own bookkeeping on a fresh read, and a
|
|
3069
|
+
* superseded snapshot leaves the other writer's stamp alone), `warm`, then
|
|
3070
|
+
* the housekeeping that is never a lifecycle flip: the finished-ref
|
|
3071
|
+
* reclamation the prune already informed (item 45) and the disk sample
|
|
3072
|
+
* (item 55). */
|
|
3073
|
+
private async refreshComplete(
|
|
3074
|
+
resource: string,
|
|
3075
|
+
facts: RepoFacts,
|
|
3076
|
+
cycle: {
|
|
3077
|
+
sha: string;
|
|
3078
|
+
lockfileHash: string;
|
|
3079
|
+
committed: boolean;
|
|
3080
|
+
mintError: string | undefined;
|
|
3081
|
+
token: string | null;
|
|
3082
|
+
},
|
|
3083
|
+
): Promise<void> {
|
|
3084
|
+
const fresh = (await this.ctx.storage.get<RepoFacts>(FACTS_KEY)) ?? facts;
|
|
3085
|
+
const updatedFacts: RepoFacts = {
|
|
3086
|
+
...fresh,
|
|
3087
|
+
...(cycle.committed
|
|
3088
|
+
? { sha: cycle.sha, lockfileHash: cycle.lockfileHash, lastRefreshAt: new Date(systemClock()).toISOString() }
|
|
3089
|
+
: {}),
|
|
3090
|
+
};
|
|
3091
|
+
// Clear a PRIOR cycle's error; keep THIS cycle's mint error visible.
|
|
3092
|
+
delete updatedFacts.lastRefreshError;
|
|
3093
|
+
if (cycle.mintError) updatedFacts.lastRefreshError = cycle.mintError;
|
|
3094
|
+
// A wake cycle cleared idleSince at the gate; `facts` was read at entry and
|
|
3095
|
+
// still carries it — never resurrect it here (the dash would show a stale
|
|
3096
|
+
// "idle since" and every attach would take the wake-fetch path).
|
|
3097
|
+
delete updatedFacts.idleSince;
|
|
3098
|
+
await this.ctx.storage.put(FACTS_KEY, updatedFacts);
|
|
3099
|
+
await this.setResidentState("warm");
|
|
3100
|
+
// Event-triggered reclamation: the prune above already told the
|
|
3101
|
+
// mirror which branches died; finished refs give their worktree and
|
|
3102
|
+
// pool user back now, not at the idle TTL. Housekeeping, never a
|
|
3103
|
+
// lifecycle flip — a failure here is a log line.
|
|
3104
|
+
try {
|
|
3105
|
+
const gc = await this.reclaimFinishedRefs(resource, facts.defaultRef, cycle.token);
|
|
3106
|
+
if (gc.reclaimed.length > 0) console.log(`reclaim ${resource}: ${JSON.stringify(gc)}`);
|
|
3107
|
+
} catch (err) {
|
|
3108
|
+
console.log(`reclaim ${resource}: pass failed: ${errMsg(err)}`);
|
|
3109
|
+
}
|
|
3110
|
+
// Item 55: the cycle's disk sample — what /residents, `repo list`, the
|
|
3111
|
+
// watchdog line and the next attach admission read. Housekeeping too.
|
|
3112
|
+
await this.measureDisk().catch((err) => console.log(`disk: measure failed: ${errMsg(err)}`));
|
|
3113
|
+
}
|
|
3114
|
+
|
|
3115
|
+
/** What a cycle's throw means. A step killed from OUTSIDE (the container
|
|
3116
|
+
* replaced under it — an image-changing deploy or a container stop; a
|
|
3117
|
+
* Worker-only deploy leaves the container running and interrupts nothing)
|
|
3118
|
+
* is `refresh-interrupted`: not evidence about the repo — it never counts
|
|
3119
|
+
* toward the park streak (the entry gate) — and the chain re-arms SHORT so
|
|
3120
|
+
* the resident is warm again within a minute instead of after the full
|
|
3121
|
+
* cadence (an unclassified kill otherwise costs the resident the whole
|
|
3122
|
+
* 10-minute cadence, e.g. `degraded(build-failed: exit 143 …)` until the
|
|
3123
|
+
* next alarm); the instance throws it to the engine, whose retry re-enters
|
|
3124
|
+
* the step. Any other failure is the repo's own: `<step>-failed: …` /
|
|
3125
|
+
* `refresh-failed: …` as before. Non-StepErrors classify too — an SDK
|
|
3126
|
+
* replacement error can surface between steps — with the generic
|
|
3127
|
+
* "refresh" step, whose failure reason is the pre-existing
|
|
3128
|
+
* `refresh-failed: …` shape. A full disk is a third class: `disk-full: …`,
|
|
3129
|
+
* never serviceable, and the one failure the resident can act on itself
|
|
3130
|
+
* (recoverFromDiskFull). */
|
|
3131
|
+
private async classifyCycleError(err: unknown): Promise<RefreshFailure> {
|
|
3132
|
+
return err instanceof StepError
|
|
3133
|
+
? await this.classifyFailure(err.step, err.message)
|
|
3134
|
+
: await this.classifyFailure("refresh", errMsg(err));
|
|
3135
|
+
}
|
|
3136
|
+
|
|
3137
|
+
/** Record a cycle's failure the one way: the classified reason in the log
|
|
3138
|
+
* (a StepError logged its own output block, but a failure between steps —
|
|
3139
|
+
* an SDK error, the markers, the snapshot — reached only the state entry,
|
|
3140
|
+
* which the next cycle's failure overwrites), on the facts (the degraded
|
|
3141
|
+
* state write can be clobbered within seconds by a concurrent attach/exec
|
|
3142
|
+
* whose ensureHydrated flips the state to `restoring · rehydrating`;
|
|
3143
|
+
* `lastRefreshError` survives that race and the next completed cycle
|
|
3144
|
+
* clears it, same as a mint error), then `degraded` with the last snapshot
|
|
3145
|
+
* still serving, and the disk-full recovery when that is the cause. */
|
|
3146
|
+
private async refreshFailed(failure: RefreshFailure, selfInFlight: number): Promise<void> {
|
|
3147
|
+
console.log(`refresh: cycle failed — ${failure.reason.slice(0, 400)}`);
|
|
3148
|
+
await this.recordRefreshError(failure.reason);
|
|
3149
|
+
await this.setResidentState("degraded", failure.reason); // last snapshot keeps serving
|
|
3150
|
+
if (failure.diskFull) await this.recoverFromDiskFull(failure.reason, selfInFlight);
|
|
3151
|
+
}
|
|
3152
|
+
|
|
3153
|
+
// -- the refresh cycle as a Workflow instance (item 7) --------------------------
|
|
3154
|
+
//
|
|
3155
|
+
// `ResidentRefresh` (the Workflow entrypoint, below the DO) calls these four
|
|
3156
|
+
// methods, one per step, through the DO stub. Each runs the same phase the
|
|
3157
|
+
// alarm runs, over the same rows, so a step the engine retries re-enters
|
|
3158
|
+
// the same idempotent read-then-act method (item 22) and finds the work
|
|
3159
|
+
// done. Inputs and answers are small facts — refs, shas, keys, a path, a
|
|
3160
|
+
// word — never a payload and never a credential: the token is minted inside
|
|
3161
|
+
// the step that needs it.
|
|
3162
|
+
|
|
3163
|
+
/** Which scheduler drives this resident's refresh cycle (LIFECYCLE_KEY). */
|
|
3164
|
+
async getLifecycle(): Promise<ResidentLifecycle> {
|
|
3165
|
+
return lifecycleOf(await this.ctx.storage.get(LIFECYCLE_KEY));
|
|
3166
|
+
}
|
|
3167
|
+
|
|
3168
|
+
/** Flip the resident between the alarm chain and the Workflow instance
|
|
3169
|
+
* (admin `/debug` `lifecycle`). To `workflow`: the pending alarm is
|
|
3170
|
+
* dropped; an alarm already firing runs to its end and re-arms nothing.
|
|
3171
|
+
* Back to `alarm`: the chain is re-armed the way the watchdog re-arms a
|
|
3172
|
+
* dead one, when the resident is in a state the chain serves. */
|
|
3173
|
+
async setLifecycle(mode: ResidentLifecycle): Promise<{ lifecycle: ResidentLifecycle; refreshSchedules: number }> {
|
|
3174
|
+
const previous = await this.getLifecycle();
|
|
3175
|
+
await this.ctx.storage.put(LIFECYCLE_KEY, mode);
|
|
3176
|
+
const resource = (await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "";
|
|
3177
|
+
if (mode === "workflow") {
|
|
3178
|
+
this.deleteSchedules(REFRESH_CALLBACK);
|
|
3179
|
+
} else if (previous !== "alarm") {
|
|
3180
|
+
const { state } = await this.getStatus();
|
|
3181
|
+
const pending = await this.listSchedules(REFRESH_CALLBACK);
|
|
3182
|
+
if (state !== "onboarding" && state !== "down" && pending.length === 0) {
|
|
3183
|
+
await this.schedule(5, REFRESH_CALLBACK, resource);
|
|
3184
|
+
}
|
|
3185
|
+
}
|
|
3186
|
+
console.log(`lifecycle: ${resource} ${previous} → ${mode}`);
|
|
3187
|
+
return { lifecycle: mode, refreshSchedules: (await this.listSchedules(REFRESH_CALLBACK)).length };
|
|
3188
|
+
}
|
|
3189
|
+
|
|
3190
|
+
private async instanceRow(): Promise<RefreshInstanceRow> {
|
|
3191
|
+
return (await this.ctx.storage.get<RefreshInstanceRow>(REFRESH_INSTANCE_KEY)) ?? { instance: null, skipped: null };
|
|
3192
|
+
}
|
|
3193
|
+
|
|
3194
|
+
/** The facts the cron's instance-creation decision reads
|
|
3195
|
+
* (`shouldCreateRefreshInstance`): the flag, the state and when it last
|
|
3196
|
+
* changed, idle mode, the last instance's creation time. */
|
|
3197
|
+
async refreshRow(): Promise<RefreshRow> {
|
|
3198
|
+
const map = await this.ctx.storage.get<unknown>([
|
|
3199
|
+
LIFECYCLE_KEY,
|
|
3200
|
+
STATE_KEY,
|
|
3201
|
+
UPDATED_KEY,
|
|
3202
|
+
FACTS_KEY,
|
|
3203
|
+
REFRESH_INSTANCE_KEY,
|
|
3204
|
+
]);
|
|
3205
|
+
const facts = map.get(FACTS_KEY) as RepoFacts | undefined;
|
|
3206
|
+
const row = (map.get(REFRESH_INSTANCE_KEY) as RefreshInstanceRow | undefined) ?? { instance: null, skipped: null };
|
|
3207
|
+
const epochMs = (iso: string | undefined): number | null => {
|
|
3208
|
+
const t = Date.parse(iso ?? "");
|
|
3209
|
+
return Number.isFinite(t) ? t : null;
|
|
3210
|
+
};
|
|
3211
|
+
return {
|
|
3212
|
+
lifecycle: lifecycleOf(map.get(LIFECYCLE_KEY)),
|
|
3213
|
+
state: (map.get(STATE_KEY) as ResidentState | undefined) ?? "down",
|
|
3214
|
+
updatedAt: epochMs(map.get(UPDATED_KEY) as string | undefined),
|
|
3215
|
+
idleSince: epochMs(facts?.idleSince),
|
|
3216
|
+
lastInstanceAt: epochMs(row.instance?.createdAt),
|
|
3217
|
+
instanceRunning: await this.instanceRunning(row.instance?.id ?? null),
|
|
3218
|
+
};
|
|
3219
|
+
}
|
|
3220
|
+
|
|
3221
|
+
/** Whether the engine still runs `id`: queued, running, paused or waiting.
|
|
3222
|
+
* The one fact the marker's age cannot give — a step between retry
|
|
3223
|
+
* attempts holds no lease and writes nothing — read from the engine, which
|
|
3224
|
+
* knows. An unknown id, a missing binding or a failed read answer false:
|
|
3225
|
+
* the marker's age then decides, as before. */
|
|
3226
|
+
private async instanceRunning(id: string | null): Promise<boolean> {
|
|
3227
|
+
if (!id) return false;
|
|
3228
|
+
try {
|
|
3229
|
+
const { status } = await (await this.env.RESIDENT_REFRESH.get(id)).status();
|
|
3230
|
+
return (
|
|
3231
|
+
status === "queued" ||
|
|
3232
|
+
status === "running" ||
|
|
3233
|
+
status === "paused" ||
|
|
3234
|
+
status === "waiting" ||
|
|
3235
|
+
status === "waitingForPause"
|
|
3236
|
+
);
|
|
3237
|
+
} catch {
|
|
3238
|
+
return false;
|
|
3239
|
+
}
|
|
3240
|
+
}
|
|
3241
|
+
|
|
3242
|
+
/** The cron created an instance for this resident. */
|
|
3243
|
+
async recordRefreshInstance(id: string, createdAtMs: number): Promise<void> {
|
|
3244
|
+
const row = await this.instanceRow();
|
|
3245
|
+
await this.ctx.storage.put(REFRESH_INSTANCE_KEY, {
|
|
3246
|
+
...row,
|
|
3247
|
+
instance: { id, createdAt: new Date(createdAtMs).toISOString(), lastStep: null, holder: null },
|
|
3248
|
+
} satisfies RefreshInstanceRow);
|
|
3249
|
+
}
|
|
3250
|
+
|
|
3251
|
+
/** The cron did not create for this bucket: a live cycle, or the engine's duplicate refusal. */
|
|
3252
|
+
async recordRefreshSkipped(id: string, atMs: number, why: string): Promise<void> {
|
|
3253
|
+
const row = await this.instanceRow();
|
|
3254
|
+
await this.ctx.storage.put(REFRESH_INSTANCE_KEY, {
|
|
3255
|
+
...row,
|
|
3256
|
+
skipped: { id, at: new Date(atMs).toISOString(), why },
|
|
3257
|
+
} satisfies RefreshInstanceRow);
|
|
3258
|
+
}
|
|
3259
|
+
|
|
3260
|
+
/** The lifecycle flag and the instance row, for `/status` and `/debug info`. */
|
|
3261
|
+
async getRefreshView(): Promise<{
|
|
3262
|
+
lifecycle: ResidentLifecycle;
|
|
3263
|
+
instance: { id: string; createdAt: string; lastStep: string | null } | null;
|
|
3264
|
+
skipped: RefreshInstanceRow["skipped"];
|
|
3265
|
+
}> {
|
|
3266
|
+
const [lifecycle, row] = await Promise.all([this.getLifecycle(), this.instanceRow()]);
|
|
3267
|
+
const instance = row.instance
|
|
3268
|
+
? { id: row.instance.id, createdAt: row.instance.createdAt, lastStep: row.instance.lastStep }
|
|
3269
|
+
: null;
|
|
3270
|
+
return { lifecycle, instance, skipped: row.skipped };
|
|
3271
|
+
}
|
|
3272
|
+
|
|
3273
|
+
/** An instance step ended: its outcome on the row, for the operator. An
|
|
3274
|
+
* instance the row does not know (created by hand) is adopted. */
|
|
3275
|
+
private async recordInstanceStep(instance: string, step: string, outcome: string): Promise<void> {
|
|
3276
|
+
const row = await this.instanceRow();
|
|
3277
|
+
const known = row.instance?.id === instance ? row.instance : null;
|
|
3278
|
+
await this.ctx.storage.put(REFRESH_INSTANCE_KEY, {
|
|
3279
|
+
...row,
|
|
3280
|
+
instance: {
|
|
3281
|
+
id: instance,
|
|
3282
|
+
createdAt: known?.createdAt ?? new Date(systemClock()).toISOString(),
|
|
3283
|
+
holder: known?.holder ?? null,
|
|
3284
|
+
lastStep: residentText(`${step}: ${outcome}`),
|
|
3285
|
+
},
|
|
3286
|
+
} satisfies RefreshInstanceRow);
|
|
3287
|
+
}
|
|
3288
|
+
|
|
3289
|
+
/** The instance's fetch step took the cycle lease: remember the holder so a
|
|
3290
|
+
* later step — in this incarnation or the next — can release exactly it. */
|
|
3291
|
+
private async recordInstanceHolder(instance: string, holder: string): Promise<void> {
|
|
3292
|
+
const row = await this.instanceRow();
|
|
3293
|
+
const known = row.instance?.id === instance ? row.instance : null;
|
|
3294
|
+
await this.ctx.storage.put(REFRESH_INSTANCE_KEY, {
|
|
3295
|
+
...row,
|
|
3296
|
+
instance: {
|
|
3297
|
+
id: instance,
|
|
3298
|
+
createdAt: known?.createdAt ?? new Date(systemClock()).toISOString(),
|
|
3299
|
+
lastStep: known?.lastStep ?? null,
|
|
3300
|
+
holder,
|
|
3301
|
+
},
|
|
3302
|
+
} satisfies RefreshInstanceRow);
|
|
3303
|
+
}
|
|
3304
|
+
|
|
3305
|
+
/** Release the cycle lease the instance holds, if any. */
|
|
3306
|
+
private async clearInstanceLease(instance: string): Promise<void> {
|
|
3307
|
+
const row = await this.instanceRow();
|
|
3308
|
+
if (row.instance?.id !== instance || !row.instance.holder) return;
|
|
3309
|
+
await this.clearInFlight("refresh", row.instance.holder);
|
|
3310
|
+
await this.ctx.storage.put(REFRESH_INSTANCE_KEY, {
|
|
3311
|
+
...row,
|
|
3312
|
+
instance: { ...row.instance, holder: null },
|
|
3313
|
+
} satisfies RefreshInstanceRow);
|
|
3314
|
+
}
|
|
3315
|
+
|
|
3316
|
+
/** Run one step of the refresh instance the way the alarm runs its cycle:
|
|
3317
|
+
* counted in flight (so an attach-path reconcileImage never stops the
|
|
3318
|
+
* container under it), under a step trace the instance grafts on its root,
|
|
3319
|
+
* its outcome on the instance row for `/status`. A step killed from outside
|
|
3320
|
+
* (the container replaced under it) is thrown to the engine, whose retry
|
|
3321
|
+
* re-enters the same idempotent method — the row stays `refreshing`, never
|
|
3322
|
+
* `degraded`, and a `refreshing` younger than the stale bound keeps the
|
|
3323
|
+
* cron from creating a second instance meanwhile. A failure of the repo's
|
|
3324
|
+
* own is recorded as the alarm records it — `degraded` with the reason,
|
|
3325
|
+
* the last snapshot still serving — and answered `failed`, which ends the
|
|
3326
|
+
* instance; the next cron firing starts the next cycle from that state. */
|
|
3327
|
+
private async runInstanceStep<T>(
|
|
3328
|
+
instance: string,
|
|
3329
|
+
step: string,
|
|
3330
|
+
fn: () => Promise<InstanceStepResult<T>>,
|
|
3331
|
+
): Promise<InstanceStepAnswer<T>> {
|
|
3332
|
+
const startedAt = systemClock();
|
|
3333
|
+
const trace = createStepTrace(startedAt);
|
|
3334
|
+
this.refreshesInFlight++;
|
|
3335
|
+
let outcome = "done";
|
|
3336
|
+
try {
|
|
3337
|
+
const result = await this.stepTrace.run(trace, fn);
|
|
3338
|
+
if (result.status !== "done") {
|
|
3339
|
+
outcome = result.status === "stopped" ? `stopped (${result.why})` : `failed (${result.reason})`;
|
|
3340
|
+
await this.clearInstanceLease(instance);
|
|
3341
|
+
}
|
|
3342
|
+
return { ...result, startedAt, trace: trace.steps() };
|
|
3343
|
+
} catch (err) {
|
|
3344
|
+
if (err instanceof ResidentDownError) {
|
|
3345
|
+
// Already down with its reason (goDown recorded it); the instance ends.
|
|
3346
|
+
outcome = `failed (${err.message})`;
|
|
3347
|
+
await this.clearInstanceLease(instance);
|
|
3348
|
+
return { status: "failed", reason: err.message, startedAt, trace: trace.steps() };
|
|
3349
|
+
}
|
|
3350
|
+
const failure = await this.classifyCycleError(err);
|
|
3351
|
+
if (failure.interrupted) {
|
|
3352
|
+
outcome = `interrupted (${failure.reason}) — the engine retries`;
|
|
3353
|
+
console.log(`refresh instance ${instance}: ${step} interrupted — ${failure.reason.slice(0, 400)}; retrying`);
|
|
3354
|
+
throw err;
|
|
3355
|
+
}
|
|
3356
|
+
await this.refreshFailed(failure, 1);
|
|
3357
|
+
await this.clearInstanceLease(instance);
|
|
3358
|
+
outcome = `failed (${failure.reason})`;
|
|
3359
|
+
return { status: "failed", reason: failure.reason, startedAt, trace: trace.steps() };
|
|
3360
|
+
} finally {
|
|
3361
|
+
this.refreshesInFlight--;
|
|
3362
|
+
// The gates set this for the alarm's finally; no alarm runs on this path.
|
|
3363
|
+
this.rearmOutcome = "normal";
|
|
3364
|
+
await this.recordInstanceStep(instance, step, outcome).catch((err) =>
|
|
3365
|
+
console.log(`refresh instance ${instance}: recording ${step} failed: ${errMsg(err)}`),
|
|
3366
|
+
);
|
|
2575
3367
|
}
|
|
2576
3368
|
}
|
|
2577
3369
|
|
|
3370
|
+
/** Step `fetch`: the gates, the cycle lease, the fetch and the plan. The
|
|
3371
|
+
* instance id is the cycle `fetchMirror` records, so a retry of this step
|
|
3372
|
+
* finds its fetch done. A rebuild's markers come off here, once per cycle,
|
|
3373
|
+
* never at the build step — a retried build must find its own work done. */
|
|
3374
|
+
async refreshInstanceFetch(input: {
|
|
3375
|
+
resource: string;
|
|
3376
|
+
instance: string;
|
|
3377
|
+
}): Promise<InstanceStepAnswer<RefreshFetchFacts>> {
|
|
3378
|
+
return this.runInstanceStep<RefreshFetchFacts>(input.instance, "fetch", async () => {
|
|
3379
|
+
const before = await this.getStatus();
|
|
3380
|
+
// down stays down (a rebuild is the escape hatch); onboarding is owned by provisioning.
|
|
3381
|
+
if (before.state === "onboarding" || before.state === "down") return { status: "stopped", why: "not-serving" };
|
|
3382
|
+
const gate = await this.refreshGate(input.resource);
|
|
3383
|
+
if (!gate.go) return { status: "stopped", why: gate.why };
|
|
3384
|
+
const { record, facts } = gate;
|
|
3385
|
+
const holder = this.nextHolder();
|
|
3386
|
+
await this.recordInFlight("refresh", holder, REFRESH_CYCLE_LEASE_MS, "refresh");
|
|
3387
|
+
await this.recordInstanceHolder(input.instance, holder);
|
|
3388
|
+
const fetched = await this.refreshFetch(input.resource, facts, input.instance, 1);
|
|
3389
|
+
if (!fetched.ok) return { status: "failed", reason: fetched.reason };
|
|
3390
|
+
const { sha, lockfileHash } = fetched;
|
|
3391
|
+
const plan = planRefresh({
|
|
3392
|
+
sha,
|
|
3393
|
+
factsSha: facts.sha,
|
|
3394
|
+
lockfileKey: lockfileHash,
|
|
3395
|
+
disk: await this.readRefreshDisk(),
|
|
3396
|
+
});
|
|
3397
|
+
if (plan.action !== "unchanged") {
|
|
3398
|
+
console.log(`refresh: ${facts.sha.slice(0, 8)} → ${sha.slice(0, 8)}: ${plan.action} (${plan.why})`);
|
|
3399
|
+
}
|
|
3400
|
+
if (plan.action === "rebuild") await this.refreshClearMarkers(plan.install);
|
|
3401
|
+
return {
|
|
3402
|
+
status: "done",
|
|
3403
|
+
ref: facts.defaultRef,
|
|
3404
|
+
sha,
|
|
3405
|
+
factsSha: facts.sha,
|
|
3406
|
+
lockfileKey: lockfileHash,
|
|
3407
|
+
action: plan.action,
|
|
3408
|
+
install: plan.action === "rebuild" && plan.install && !!record.commands.install,
|
|
3409
|
+
mintError: fetched.mintError ?? null,
|
|
3410
|
+
};
|
|
3411
|
+
});
|
|
3412
|
+
}
|
|
3413
|
+
|
|
3414
|
+
/** Step `install`: the store entry for the new key (`installDeps` finds a
|
|
3415
|
+
* complete entry done). Answers the entry's path for the build to link. */
|
|
3416
|
+
async refreshInstanceInstall(input: {
|
|
3417
|
+
resource: string;
|
|
3418
|
+
instance: string;
|
|
3419
|
+
sha: string;
|
|
3420
|
+
lockfileKey: string;
|
|
3421
|
+
}): Promise<InstanceStepAnswer<{ entry: string | null }>> {
|
|
3422
|
+
return this.runInstanceStep<{ entry: string | null }>(input.instance, "install", async () => {
|
|
3423
|
+
const record = await this.registry().getRecord(input.resource);
|
|
3424
|
+
if (!record) return { status: "stopped", why: "offboarded" };
|
|
3425
|
+
const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
|
|
3426
|
+
if (!facts) return { status: "stopped", why: "no-facts" };
|
|
3427
|
+
const entry = await this.refreshInstall(record, facts, input.sha, input.lockfileKey);
|
|
3428
|
+
return { status: "done", entry };
|
|
3429
|
+
});
|
|
3430
|
+
}
|
|
3431
|
+
|
|
3432
|
+
/** Step `build`: the checkout to the sha and its build (`runBuild` finds a
|
|
3433
|
+
* checkout whose markers all name the target done). */
|
|
3434
|
+
async refreshInstanceBuild(input: {
|
|
3435
|
+
resource: string;
|
|
3436
|
+
instance: string;
|
|
3437
|
+
sha: string;
|
|
3438
|
+
factsSha: string;
|
|
3439
|
+
lockfileKey: string;
|
|
3440
|
+
depsEntry: string | null;
|
|
3441
|
+
}): Promise<InstanceStepAnswer<{ ran: boolean; why: string }>> {
|
|
3442
|
+
return this.runInstanceStep<{ ran: boolean; why: string }>(input.instance, "build", async () => {
|
|
3443
|
+
const record = await this.registry().getRecord(input.resource);
|
|
3444
|
+
if (!record) return { status: "stopped", why: "offboarded" };
|
|
3445
|
+
const built = await this.runBuild({
|
|
3446
|
+
sha: input.sha,
|
|
3447
|
+
factsSha: input.factsSha,
|
|
3448
|
+
lockfileKey: input.lockfileKey,
|
|
3449
|
+
buildCmd: record.commands.build,
|
|
3450
|
+
depsEntry: input.depsEntry,
|
|
3451
|
+
});
|
|
3452
|
+
// `done` from the plan means the tree was already built for this sha; the
|
|
3453
|
+
// step reports whether a build actually ran.
|
|
3454
|
+
return { status: "done", ran: !built.done, why: built.why };
|
|
3455
|
+
});
|
|
3456
|
+
}
|
|
3457
|
+
|
|
3458
|
+
/** Step `snapshot`: the stamped pair to R2 (`snapshot` finds a record at the
|
|
3459
|
+
* stamp done and answers `superseded` to another writer, never a throw),
|
|
3460
|
+
* then the cycle's completion — facts, `warm`, the reclamation, the disk
|
|
3461
|
+
* sample — and the cycle lease released. An `unchanged` cycle skips the
|
|
3462
|
+
* archive and still completes, as the alarm does. */
|
|
3463
|
+
async refreshInstanceSnapshot(input: {
|
|
3464
|
+
resource: string;
|
|
3465
|
+
instance: string;
|
|
3466
|
+
ref: string;
|
|
3467
|
+
sha: string;
|
|
3468
|
+
lockfileKey: string;
|
|
3469
|
+
action: RefreshPlan["action"];
|
|
3470
|
+
mintError: string | null;
|
|
3471
|
+
}): Promise<InstanceStepAnswer<{ committed: boolean }>> {
|
|
3472
|
+
return this.runInstanceStep<{ committed: boolean }>(input.instance, "snapshot", async () => {
|
|
3473
|
+
const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
|
|
3474
|
+
if (!facts) return { status: "stopped", why: "no-facts" };
|
|
3475
|
+
const committed =
|
|
3476
|
+
input.action === "unchanged"
|
|
3477
|
+
? true
|
|
3478
|
+
: await this.refreshSnapshot(input.resource, {
|
|
3479
|
+
ref: input.ref,
|
|
3480
|
+
sha: input.sha,
|
|
3481
|
+
lockfileHash: input.lockfileKey,
|
|
3482
|
+
});
|
|
3483
|
+
// The reclamation's token: minted here (cached per slug), never carried
|
|
3484
|
+
// between steps. A failed mint runs the reclamation anonymously (PR lookups
|
|
3485
|
+
// answer unknown, trees are kept) and is recorded like the fetch's.
|
|
3486
|
+
let token: string | null = null;
|
|
3487
|
+
let snapshotMintError: string | undefined;
|
|
3488
|
+
if (githubAppConfigured(this.env)) {
|
|
3489
|
+
try {
|
|
3490
|
+
token = (await mintRepoScopedToken(this.env, input.resource.slice("repo:".length))).token;
|
|
3491
|
+
} catch (err) {
|
|
3492
|
+
snapshotMintError = `token-mint-failed (reclamation runs anonymously): ${errMsg(err)}`;
|
|
3493
|
+
console.log(`refresh instance ${input.instance}: ${snapshotMintError}`);
|
|
3494
|
+
}
|
|
3495
|
+
}
|
|
3496
|
+
await this.refreshComplete(input.resource, facts, {
|
|
3497
|
+
sha: input.sha,
|
|
3498
|
+
lockfileHash: input.lockfileKey,
|
|
3499
|
+
committed,
|
|
3500
|
+
mintError: input.mintError ?? snapshotMintError,
|
|
3501
|
+
token,
|
|
3502
|
+
});
|
|
3503
|
+
await this.clearInstanceLease(input.instance);
|
|
3504
|
+
return { status: "done", committed };
|
|
3505
|
+
});
|
|
3506
|
+
}
|
|
3507
|
+
|
|
2578
3508
|
/** Set during one alarm by the idle gate (`idle`), an image-stale container
|
|
2579
3509
|
* stop (`image-stale-restart`), a disk-full recycle (`disk-full-restart`)
|
|
2580
3510
|
* or an interrupted step (`interrupted`) so `finally` picks the matching
|
|
@@ -2658,7 +3588,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2658
3588
|
);
|
|
2659
3589
|
await this.ctx.storage.put(DISK_FULL_RECYCLE_KEY, systemClock());
|
|
2660
3590
|
await this.recordRefreshError(`${reason} — container recycled; restoring from R2 on the next alarm`);
|
|
2661
|
-
this.
|
|
3591
|
+
this.swapIncarnation(); // deliberate incarnation swap
|
|
2662
3592
|
await this.stop().catch((err) => console.log(`disk-full: stop failed: ${errMsg(err)}`));
|
|
2663
3593
|
this.rearmOutcome = "disk-full-restart";
|
|
2664
3594
|
}
|
|
@@ -2941,7 +3871,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2941
3871
|
console.log(
|
|
2942
3872
|
`image-stale (${where}): ${last} missing in the running container — stopping so it restarts on the current image`,
|
|
2943
3873
|
);
|
|
2944
|
-
this.
|
|
3874
|
+
this.swapIncarnation(); // deliberate incarnation swap
|
|
2945
3875
|
await this.stop().catch((err) => console.log(`image-stale: stop failed: ${errMsg(err)}`));
|
|
2946
3876
|
return true;
|
|
2947
3877
|
}
|
|
@@ -2962,9 +3892,11 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2962
3892
|
action: "none" | "rearmed" | "provision-timed-out" | "auto-rebuilt";
|
|
2963
3893
|
/** Item 55: the last disk sample's gauge, for the watchdog's status line. */
|
|
2964
3894
|
disk: { usedKiB: number; totalKiB: number; freeKiB: number; at: string } | null;
|
|
3895
|
+
/** Item 7: what the cron's instance-creation decision reads, after the check above settled the state. */
|
|
3896
|
+
refresh: RefreshRow;
|
|
2965
3897
|
}> {
|
|
2966
3898
|
const [check, disk] = await Promise.all([this.watchdogCheckLifecycle(), this.diskGauge()]);
|
|
2967
|
-
return { ...check, disk };
|
|
3899
|
+
return { ...check, disk, refresh: await this.refreshRow() };
|
|
2968
3900
|
}
|
|
2969
3901
|
|
|
2970
3902
|
private async watchdogCheckLifecycle(): Promise<{
|
|
@@ -3007,6 +3939,10 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3007
3939
|
// Any serving state clears accumulated strikes (a recovery must reset the
|
|
3008
3940
|
// counter, or an unrelated later down inherits stale strikes).
|
|
3009
3941
|
await this.ctx.storage.delete(REBUILD_STRIKES_KEY);
|
|
3942
|
+
// Item 7: a resident on the Workflow lifecycle has no chain to re-arm —
|
|
3943
|
+
// its cycles are the instances the cron creates. The sweep re-arm below
|
|
3944
|
+
// is housekeeping either way; the two refresh re-arms are the alarm's.
|
|
3945
|
+
const chained = (await this.getLifecycle()) === "alarm";
|
|
3010
3946
|
|
|
3011
3947
|
// The sweep chain has the same failure mode as the refresh chain (a DO
|
|
3012
3948
|
// eviction mid-callback kills the self-rescheduling), but nothing re-armed
|
|
@@ -3050,28 +3986,49 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3050
3986
|
// named degradation, never a stall) and pull the next cycle to +5s so it normalizes.
|
|
3051
3987
|
if (status.state === "refreshing" || status.state === "restoring") {
|
|
3052
3988
|
const updatedAt = Date.parse((await this.ctx.storage.get<string>(UPDATED_KEY)) ?? "") || 0;
|
|
3053
|
-
//
|
|
3054
|
-
//
|
|
3055
|
-
//
|
|
3056
|
-
//
|
|
3057
|
-
//
|
|
3989
|
+
// Who holds what comes from the in-flight row (item 22), not from this
|
|
3990
|
+
// isolate's memory: a cycle or hydration lease is alive only for the
|
|
3991
|
+
// current incarnation and inside its budget. A hydration past the stale
|
|
3992
|
+
// bound counts as DEAD, not in flight: its promise lives on SDK calls
|
|
3993
|
+
// into a container that may have been replaced under it, and a promise
|
|
3994
|
+
// that never settles would otherwise hold the memo forever — making a
|
|
3995
|
+
// stuck `restoring` permanently invisible to this branch. No legitimate
|
|
3058
3996
|
// restore approaches STALE_MIDFLIGHT_MS (a full R2 restore is ~1 min).
|
|
3059
|
-
const
|
|
3060
|
-
const inFlight =
|
|
3997
|
+
const live = liveInFlight(await this.readInFlight(), systemClock(), this.incarnation);
|
|
3998
|
+
const inFlight = live.refresh || live.hydration;
|
|
3061
3999
|
if (!inFlight && systemClock() - updatedAt > STALE_MIDFLIGHT_MS) {
|
|
3062
4000
|
// The reads above yielded; a cycle that started meanwhile owns the
|
|
3063
4001
|
// state now — leave it alone rather than stamp `degraded` over it.
|
|
3064
4002
|
const again = await this.getStatus();
|
|
3065
|
-
const
|
|
3066
|
-
|
|
3067
|
-
if (again.state !== status.state ||
|
|
4003
|
+
const rowAgain = await this.readInFlight();
|
|
4004
|
+
const liveAgain = liveInFlight(rowAgain, systemClock(), this.incarnation);
|
|
4005
|
+
if (again.state !== status.state || liveAgain.refresh || liveAgain.hydration) {
|
|
3068
4006
|
return { resource, ...again, action: "none" };
|
|
3069
4007
|
}
|
|
3070
|
-
// Drop the dead hydration reference
|
|
3071
|
-
// ensureHydrated starts a fresh restore instead of
|
|
3072
|
-
// that will never settle. Safe: past the bound
|
|
3073
|
-
// end is still writing (the container it talked
|
|
4008
|
+
// Drop the dead hydration reference and the dead leases so the
|
|
4009
|
+
// re-armed cycle's ensureHydrated starts a fresh restore instead of
|
|
4010
|
+
// awaiting a promise that will never settle. Safe: past the bound
|
|
4011
|
+
// nothing on the other end is still writing (the container it talked
|
|
4012
|
+
// to is gone). Each clear is compared against the holder just read:
|
|
4013
|
+
// a cycle that recorded a fresh lease between that read and this
|
|
4014
|
+
// delete keeps it, the way a release never deletes another holder's row.
|
|
3074
4015
|
this.hydration = null;
|
|
4016
|
+
if (!chained && (await this.instanceRunning((await this.instanceRow()).instance?.id ?? null))) {
|
|
4017
|
+
// The engine still runs the recorded instance — a step between retry
|
|
4018
|
+
// attempts, holding no lease and writing no state. Not stale: leave
|
|
4019
|
+
// the marker, create nothing (the cron's decision reads the same fact).
|
|
4020
|
+
return { resource, ...status, action: "none" };
|
|
4021
|
+
}
|
|
4022
|
+
if (rowAgain.refresh) await this.clearInFlight("refresh", rowAgain.refresh.holder);
|
|
4023
|
+
if (rowAgain.hydration) await this.clearInFlight("hydration", rowAgain.hydration.holder);
|
|
4024
|
+
if (!chained) {
|
|
4025
|
+
// The orphan is named the same way; the next instance the cron
|
|
4026
|
+
// creates (this very pass — the marker is no longer `refreshing`)
|
|
4027
|
+
// normalizes it, no alarm involved.
|
|
4028
|
+
const reason = `stale-mid-flight: ${status.state} since ${new Date(updatedAt).toISOString()} with no cycle running; the next refresh instance normalizes it`;
|
|
4029
|
+
await this.setResidentState("degraded", reason);
|
|
4030
|
+
return { resource, state: "degraded", reason, action: "none" };
|
|
4031
|
+
}
|
|
3075
4032
|
const reason = `stale-mid-flight: ${status.state} since ${new Date(updatedAt).toISOString()} with no cycle running; re-armed by watchdog`;
|
|
3076
4033
|
await this.setResidentState("degraded", reason);
|
|
3077
4034
|
this.deleteSchedules(REFRESH_CALLBACK);
|
|
@@ -3080,6 +4037,8 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3080
4037
|
}
|
|
3081
4038
|
}
|
|
3082
4039
|
|
|
4040
|
+
// No chain to be dead under the Workflow lifecycle (item 7).
|
|
4041
|
+
if (!chained) return { resource, ...status, action: "none" };
|
|
3083
4042
|
const pending = await this.listSchedules(REFRESH_CALLBACK);
|
|
3084
4043
|
if (pending.length === 0) {
|
|
3085
4044
|
await this.schedule(5, REFRESH_CALLBACK, resource);
|
|
@@ -3788,8 +4747,9 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3788
4747
|
* `restoring` span stays under RESTORE_MAX_MS in total (the hydrate
|
|
3789
4748
|
* invariant); every other caller gets `min(budgetMs, RESTORE_MAX_MS)`
|
|
3790
4749
|
* from now, so an attach never waits longer for a download than it
|
|
3791
|
-
* would for an install.
|
|
3792
|
-
|
|
4750
|
+
* would for an install. `attempt`: the private scratch tree's name when
|
|
4751
|
+
* the caller has leased it (installDeps); minted here otherwise. */
|
|
4752
|
+
opts: { seedFromKey?: string; restoreDeadlineMs?: number; attempt?: string } = {},
|
|
3793
4753
|
): Promise<string> {
|
|
3794
4754
|
const backupRecord = await this.depsBackupRecord(key);
|
|
3795
4755
|
const plan = planDepsMaterialization({
|
|
@@ -3808,14 +4768,15 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3808
4768
|
// same way next time) and falls through to the installer, which records
|
|
3809
4769
|
// a fresh backup — never a stranded key.
|
|
3810
4770
|
const restoreDeadlineMs = opts.restoreDeadlineMs ?? systemClock() + Math.min(budgetMs, RESTORE_MAX_MS);
|
|
4771
|
+
const attempt = opts.attempt ?? crypto.randomUUID().slice(0, 8);
|
|
3811
4772
|
const p = (
|
|
3812
4773
|
plan.action === "restore" && backupRecord
|
|
3813
|
-
? this.restoreDepsEntry(key, backupRecord, restoreDeadlineMs).catch(async (err) => {
|
|
4774
|
+
? this.restoreDepsEntry(key, backupRecord, restoreDeadlineMs, attempt).catch(async (err) => {
|
|
3814
4775
|
console.log(`deps: restore of ${key.slice(0, 8)} failed — installing instead: ${errMsg(err)}`);
|
|
3815
4776
|
await this.dropDepsBackups([key]).catch(() => {});
|
|
3816
|
-
return this.installDepsEntry(key, sha, installCmd, budgetMs, opts);
|
|
4777
|
+
return this.installDepsEntry(key, sha, installCmd, budgetMs, { ...opts, attempt });
|
|
3817
4778
|
})
|
|
3818
|
-
: this.installDepsEntry(key, sha, installCmd, budgetMs, opts)
|
|
4779
|
+
: this.installDepsEntry(key, sha, installCmd, budgetMs, { ...opts, attempt })
|
|
3819
4780
|
).finally(() => this.depsInFlight.delete(key));
|
|
3820
4781
|
this.depsInFlight.set(key, p);
|
|
3821
4782
|
return p;
|
|
@@ -3875,8 +4836,12 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3875
4836
|
* then the same commit script an install ends with: staging, atomic
|
|
3876
4837
|
* rename, `.complete` LAST. A partial download never becomes an entry.
|
|
3877
4838
|
* `deadlineMs` is the caller's (see materializeDeps): never a fresh cap. */
|
|
3878
|
-
private async restoreDepsEntry(
|
|
3879
|
-
|
|
4839
|
+
private async restoreDepsEntry(
|
|
4840
|
+
key: string,
|
|
4841
|
+
record: DepsBackupRecord,
|
|
4842
|
+
deadlineMs: number,
|
|
4843
|
+
attempt: string,
|
|
4844
|
+
): Promise<string> {
|
|
3880
4845
|
const scratch = depsScratchPath(attempt);
|
|
3881
4846
|
const staging = depsStagingPath(key, attempt);
|
|
3882
4847
|
const t0 = systemClock();
|
|
@@ -3933,10 +4898,10 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3933
4898
|
sha: string,
|
|
3934
4899
|
installCmd: string,
|
|
3935
4900
|
budgetMs: number,
|
|
3936
|
-
opts: { seedFromKey?: string },
|
|
4901
|
+
opts: { seedFromKey?: string; attempt: string },
|
|
3937
4902
|
): Promise<string> {
|
|
3938
4903
|
await this.acquireDepsInstallSlot();
|
|
3939
|
-
const attempt =
|
|
4904
|
+
const { attempt } = opts;
|
|
3940
4905
|
const scratch = depsScratchPath(attempt);
|
|
3941
4906
|
const staging = depsStagingPath(key, attempt);
|
|
3942
4907
|
const t0 = systemClock();
|
|
@@ -5145,10 +6110,19 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5145
6110
|
FACTS_KEY,
|
|
5146
6111
|
SNAPSHOT_KEY,
|
|
5147
6112
|
DISK_KEY,
|
|
6113
|
+
MIRROR_MUTEX_KEY,
|
|
6114
|
+
inFlightKey("refresh"),
|
|
6115
|
+
inFlightKey("hydration"),
|
|
6116
|
+
LIFECYCLE_KEY,
|
|
6117
|
+
REFRESH_INSTANCE_KEY,
|
|
5148
6118
|
]);
|
|
5149
6119
|
const facts = map.get(FACTS_KEY) as RepoFacts | undefined;
|
|
5150
6120
|
const snap = map.get(SNAPSHOT_KEY) as SnapshotRecord | undefined;
|
|
5151
6121
|
const disk = (map.get(DISK_KEY) as DiskSample | undefined) ?? null;
|
|
6122
|
+
const refreshRow = (map.get(REFRESH_INSTANCE_KEY) as RefreshInstanceRow | undefined) ?? {
|
|
6123
|
+
instance: null,
|
|
6124
|
+
skipped: null,
|
|
6125
|
+
};
|
|
5152
6126
|
const [refresh, provisionRun, provisionDeadline, bindings] = await Promise.all([
|
|
5153
6127
|
this.listSchedules(REFRESH_CALLBACK),
|
|
5154
6128
|
this.listSchedules(PROVISION_RUN_CALLBACK),
|
|
@@ -5203,6 +6177,28 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5203
6177
|
inFlight: this.inFlightCount(),
|
|
5204
6178
|
// The runs alone (no refresh cycle): what the deploy preflight refuses on.
|
|
5205
6179
|
runsInFlight: this.runsInFlightCount(),
|
|
6180
|
+
// Item 22: who holds what, as the rows say — the mirror mutex and the
|
|
6181
|
+
// cycle/hydration leases, each judged against this incarnation.
|
|
6182
|
+
incarnation: this.incarnation,
|
|
6183
|
+
mirrorMutex: (map.get(MIRROR_MUTEX_KEY) as Lease | undefined) ?? null,
|
|
6184
|
+
leases: inFlightRow(
|
|
6185
|
+
map.get(inFlightKey("refresh")) as Lease | undefined,
|
|
6186
|
+
map.get(inFlightKey("hydration")) as Lease | undefined,
|
|
6187
|
+
),
|
|
6188
|
+
// Item 7: which scheduler drives the refresh cycle, and — on the
|
|
6189
|
+
// Workflow lifecycle — the instance the cron last created with the step
|
|
6190
|
+
// it last reported, and the last bucket the cron skipped.
|
|
6191
|
+
lifecycle: lifecycleOf(map.get(LIFECYCLE_KEY)),
|
|
6192
|
+
refresh: {
|
|
6193
|
+
instance: refreshRow.instance
|
|
6194
|
+
? {
|
|
6195
|
+
id: refreshRow.instance.id,
|
|
6196
|
+
createdAt: refreshRow.instance.createdAt,
|
|
6197
|
+
lastStep: refreshRow.instance.lastStep,
|
|
6198
|
+
}
|
|
6199
|
+
: null,
|
|
6200
|
+
skipped: refreshRow.skipped,
|
|
6201
|
+
},
|
|
5206
6202
|
threads,
|
|
5207
6203
|
// Item 55: the last disk sample (`residentDiskBudget.ts` DiskSample), or
|
|
5208
6204
|
// null before the first measurement of this incarnation.
|
|
@@ -5228,12 +6224,16 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5228
6224
|
return { killed: true, remaining: (await this.listSchedules(REFRESH_CALLBACK)).length };
|
|
5229
6225
|
}
|
|
5230
6226
|
|
|
5231
|
-
/** Pull the next refresh forward to ~1s from now.
|
|
5232
|
-
|
|
6227
|
+
/** Pull the next refresh forward to ~1s from now. A resident on the Workflow
|
|
6228
|
+
* lifecycle has no chain to pull: its next cycle is the instance the next
|
|
6229
|
+
* cron firing creates (`run-watchdog` runs that pass on demand). */
|
|
6230
|
+
async debugRefreshNow(): Promise<{ scheduled: boolean; lifecycle: ResidentLifecycle }> {
|
|
6231
|
+
const lifecycle = await this.getLifecycle();
|
|
6232
|
+
if (lifecycle === "workflow") return { scheduled: false, lifecycle };
|
|
5233
6233
|
const resource = (await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "";
|
|
5234
6234
|
this.deleteSchedules(REFRESH_CALLBACK);
|
|
5235
6235
|
await this.schedule(1, REFRESH_CALLBACK, resource);
|
|
5236
|
-
return { scheduled: true };
|
|
6236
|
+
return { scheduled: true, lifecycle };
|
|
5237
6237
|
}
|
|
5238
6238
|
|
|
5239
6239
|
/** Fault injection for the watchdog's stuck-onboarding path: re-persist
|
|
@@ -5250,7 +6250,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5250
6250
|
* next refresh must take the restoring→warm wake path). */
|
|
5251
6251
|
async debugStopContainer(): Promise<{ stopped: boolean; error?: string }> {
|
|
5252
6252
|
try {
|
|
5253
|
-
this.
|
|
6253
|
+
this.swapIncarnation(); // deliberate incarnation swap
|
|
5254
6254
|
await this.stop();
|
|
5255
6255
|
return { stopped: true };
|
|
5256
6256
|
} catch (err) {
|
|
@@ -5423,7 +6423,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5423
6423
|
}
|
|
5424
6424
|
// Retired DO: clear the alarm the Container base may have armed for its
|
|
5425
6425
|
// schedules, then wipe storage so nothing ever wakes this object again.
|
|
5426
|
-
this.
|
|
6426
|
+
this.swapIncarnation(); // retired object, retired memos
|
|
5427
6427
|
await this.ctx.storage.deleteAlarm();
|
|
5428
6428
|
await this.ctx.storage.deleteAll();
|
|
5429
6429
|
return { schedulesCancelled: true, containerStopped, storageCleared: true, backupObjectsDeleted, errors };
|
|
@@ -6213,13 +7213,22 @@ async function handleStatus(env: Env, url: URL): Promise<Response> {
|
|
|
6213
7213
|
// deploy gate reads /residents. The registry check rides in the same flight
|
|
6214
7214
|
// (its 404 is judged first, the probes' results discarded then).
|
|
6215
7215
|
const stub = residentStub(env, resource.resource);
|
|
6216
|
-
const [record, status, inFlight] = await Promise.all([
|
|
7216
|
+
const [record, status, inFlight, refresh] = await Promise.all([
|
|
6217
7217
|
registryStub(env).getRecord(resource.resource),
|
|
6218
7218
|
stub.getStatus(),
|
|
6219
7219
|
stub.getInFlightCount(),
|
|
7220
|
+
stub.getRefreshView(),
|
|
6220
7221
|
]);
|
|
6221
7222
|
if (!record) return json({ error: `${resource.resource} is not onboarded` }, 404);
|
|
6222
|
-
|
|
7223
|
+
// Item 7: which scheduler drives the refresh cycle and, on the Workflow
|
|
7224
|
+
// lifecycle, the current instance with its last step and the last skipped bucket.
|
|
7225
|
+
return json({
|
|
7226
|
+
state: status.state,
|
|
7227
|
+
reason: status.reason,
|
|
7228
|
+
inFlight,
|
|
7229
|
+
lifecycle: refresh.lifecycle,
|
|
7230
|
+
refresh: { instance: refresh.instance, skipped: refresh.skipped },
|
|
7231
|
+
});
|
|
6223
7232
|
}
|
|
6224
7233
|
|
|
6225
7234
|
// -- thread data plane handlers -----------------------------------------------
|
|
@@ -6578,20 +7587,87 @@ async function handleDebug(env: Env, body: Record<string, unknown>): Promise<Res
|
|
|
6578
7587
|
if ("error" in days) return json({ error: days.error }, 400);
|
|
6579
7588
|
return json(await stub.debugBackdateThread(threadKey.threadKey, days.value));
|
|
6580
7589
|
}
|
|
7590
|
+
case "lifecycle": {
|
|
7591
|
+
// Item 7: which scheduler drives this resident's refresh cycle. Admin
|
|
7592
|
+
// scope, one resident at a time; the default for every resident is `alarm`.
|
|
7593
|
+
const mode = parseLifecycle(body.mode);
|
|
7594
|
+
if (!mode) return json({ error: 'mode must be "alarm" or "workflow"' }, 400);
|
|
7595
|
+
return json({ op, resource: resource.resource, ...(await stub.setLifecycle(mode)) });
|
|
7596
|
+
}
|
|
6581
7597
|
default:
|
|
6582
7598
|
return json(
|
|
6583
7599
|
{
|
|
6584
|
-
error: `unknown op ${JSON.stringify(op)} (ops: info, schedules, kill-refresh, refresh-now, stop-container, force-onboarding, force-down, mint-token, run-watchdog, set-test-overrides, threads, sweep-now, reclaim-now, measure-disk, purge-bindings, backdate-thread)`,
|
|
7600
|
+
error: `unknown op ${JSON.stringify(op)} (ops: info, schedules, kill-refresh, refresh-now, stop-container, force-onboarding, force-down, mint-token, run-watchdog, set-test-overrides, threads, sweep-now, reclaim-now, measure-disk, purge-bindings, backdate-thread, lifecycle)`,
|
|
6585
7601
|
},
|
|
6586
7602
|
400,
|
|
6587
7603
|
);
|
|
6588
7604
|
}
|
|
6589
7605
|
}
|
|
6590
7606
|
|
|
7607
|
+
/** The cron's instance-creation duty for one resident (item 7). Only a
|
|
7608
|
+
* `workflow` row gets an instance, at most one per ten-minute bucket, never
|
|
7609
|
+
* while a cycle is live (`shouldCreateRefreshInstance`); the id is
|
|
7610
|
+
* deterministic per resident and bucket, so a second firing in one bucket
|
|
7611
|
+
* meets the engine's duplicate-id refusal, which is the expected no-op. A
|
|
7612
|
+
* skipped live cycle and a duplicate are recorded on the row for `/status`.
|
|
7613
|
+
* An `alarm` resident answers null: nothing here touches it. */
|
|
7614
|
+
/** Whether the engine knows an instance by this id, in any status. A missing
|
|
7615
|
+
* id rejects on `get` or on `status`; either way the answer is false. */
|
|
7616
|
+
async function refreshInstanceExists(env: Env, id: string): Promise<boolean> {
|
|
7617
|
+
try {
|
|
7618
|
+
await (await env.RESIDENT_REFRESH.get(id)).status();
|
|
7619
|
+
return true;
|
|
7620
|
+
} catch {
|
|
7621
|
+
return false;
|
|
7622
|
+
}
|
|
7623
|
+
}
|
|
7624
|
+
|
|
7625
|
+
async function createRefreshInstance(
|
|
7626
|
+
env: Env,
|
|
7627
|
+
stub: ReturnType<typeof residentStub>,
|
|
7628
|
+
resource: string,
|
|
7629
|
+
row: RefreshRow,
|
|
7630
|
+
): Promise<RefreshInstanceAction | null> {
|
|
7631
|
+
if (row.lifecycle !== "workflow") return null;
|
|
7632
|
+
const now = systemClock();
|
|
7633
|
+
const decision = shouldCreateRefreshInstance(row, now, {
|
|
7634
|
+
intervalS: REFRESH_INTERVAL_S,
|
|
7635
|
+
idleIntervalS: IDLE_REFRESH_INTERVAL_S,
|
|
7636
|
+
});
|
|
7637
|
+
const slug = resource.slice("repo:".length);
|
|
7638
|
+
const slash = slug.indexOf("/");
|
|
7639
|
+
const id = refreshInstanceId(slug.slice(0, slash), slug.slice(slash + 1), now);
|
|
7640
|
+
if (!decision.create) {
|
|
7641
|
+
if (decision.why === "mid-cycle" || decision.why === "running")
|
|
7642
|
+
await stub.recordRefreshSkipped(id, now, decision.why);
|
|
7643
|
+
return { id, action: "skipped", why: decision.why };
|
|
7644
|
+
}
|
|
7645
|
+
try {
|
|
7646
|
+
await env.RESIDENT_REFRESH.create({ id, params: { resource } });
|
|
7647
|
+
} catch (err) {
|
|
7648
|
+
const message = errMsg(err);
|
|
7649
|
+
// The engine refuses an id that names an instance still inside its
|
|
7650
|
+
// retention, and the refusal carries no code — so the id is asked, not
|
|
7651
|
+
// the wording: an instance that answers for it exists, and the refusal
|
|
7652
|
+
// was the duplicate it looks like. Any other failure stays a failure.
|
|
7653
|
+
if (await refreshInstanceExists(env, id)) {
|
|
7654
|
+
await stub.recordRefreshSkipped(id, now, "duplicate");
|
|
7655
|
+
return { id, action: "duplicate", why: "duplicate" };
|
|
7656
|
+
}
|
|
7657
|
+
console.error(`resident-watchdog: creating refresh instance ${id} failed — ${message}`);
|
|
7658
|
+
return { id, action: "failed", why: residentText(message) };
|
|
7659
|
+
}
|
|
7660
|
+
await stub.recordRefreshInstance(id, now);
|
|
7661
|
+
console.log(`resident-watchdog: created refresh instance ${id}`);
|
|
7662
|
+
return { id, action: "created", why: decision.why };
|
|
7663
|
+
}
|
|
7664
|
+
|
|
6591
7665
|
/** One watchdog pass over every registered resident. Shared by the cron
|
|
6592
7666
|
* handler and the /debug run-watchdog op. Each check targets a different DO,
|
|
6593
7667
|
* so they run concurrently; a failing one becomes its own {error} entry
|
|
6594
|
-
* without touching its neighbors, and the results follow the registry list.
|
|
7668
|
+
* without touching its neighbors, and the results follow the registry list.
|
|
7669
|
+
* For a resident on the Workflow lifecycle the pass also creates the refresh
|
|
7670
|
+
* instance the bucket is due (item 7). */
|
|
6595
7671
|
async function runWatchdog(env: Env, parent?: TraceSpan): Promise<WatchdogSummary> {
|
|
6596
7672
|
const registry = registryStub(env);
|
|
6597
7673
|
const residents = await registry.list();
|
|
@@ -6599,13 +7675,15 @@ async function runWatchdog(env: Env, parent?: TraceSpan): Promise<WatchdogSummar
|
|
|
6599
7675
|
// one (the cron path; the /debug op runs bare), ending with the action taken
|
|
6600
7676
|
// — never the resource, which names a repo.
|
|
6601
7677
|
const checkOne = async (record: { resource: string }, span?: TraceSpan) => {
|
|
6602
|
-
const
|
|
7678
|
+
const stub = residentStub(env, record.resource);
|
|
7679
|
+
const check = await stub.watchdogCheck();
|
|
6603
7680
|
if (check.action === "provision-timed-out") {
|
|
6604
7681
|
// The DO already tried to release its own slot; this is the backstop.
|
|
6605
7682
|
await registry.remove(record.resource);
|
|
6606
7683
|
}
|
|
7684
|
+
const instance = await createRefreshInstance(env, stub, record.resource, check.refresh);
|
|
6607
7685
|
span?.setAttrs({ outcome: check.action });
|
|
6608
|
-
return check;
|
|
7686
|
+
return { ...check, instance };
|
|
6609
7687
|
};
|
|
6610
7688
|
const settled = await Promise.allSettled(
|
|
6611
7689
|
residents.map((record) =>
|
|
@@ -6621,12 +7699,141 @@ async function runWatchdog(env: Env, parent?: TraceSpan): Promise<WatchdogSummar
|
|
|
6621
7699
|
reason: s.value.reason,
|
|
6622
7700
|
action: s.value.action,
|
|
6623
7701
|
disk: s.value.disk,
|
|
7702
|
+
lifecycle: s.value.refresh.lifecycle,
|
|
7703
|
+
instance: s.value.instance,
|
|
6624
7704
|
}
|
|
6625
7705
|
: { resource: record.resource, error: errMsg(s.reason) };
|
|
6626
7706
|
});
|
|
6627
7707
|
return { cap: (await registry.limits()).cap, count: residents.length, results };
|
|
6628
7708
|
}
|
|
6629
7709
|
|
|
7710
|
+
// ---------------------------------------------------------------------------
|
|
7711
|
+
// The refresh cycle as a Workflow instance (docs/reference/specs/resident-repos.md item 7)
|
|
7712
|
+
// ---------------------------------------------------------------------------
|
|
7713
|
+
|
|
7714
|
+
/** What an instance answers when it ends: small facts for the engine's record. */
|
|
7715
|
+
interface RefreshInstanceSummary {
|
|
7716
|
+
instance: string;
|
|
7717
|
+
/** `ok`, or the word a gate or a failure ended the cycle with. */
|
|
7718
|
+
outcome: string;
|
|
7719
|
+
step: "fetch" | "install" | "build" | "snapshot";
|
|
7720
|
+
action?: RefreshPlan["action"];
|
|
7721
|
+
sha?: string;
|
|
7722
|
+
}
|
|
7723
|
+
|
|
7724
|
+
/** One refresh cycle as one short Workflow instance: `fetch`, `install` (only
|
|
7725
|
+
* when the plan moved the lockfile key), `build` (only when the branch
|
|
7726
|
+
* moved), `snapshot` — each a `step.do` calling the resident's own step
|
|
7727
|
+
* method through the DO stub, under the retry policy `REFRESH_STEP_RETRIES`
|
|
7728
|
+
* (six attempts, thirty seconds apart, doubling: about 15.5 minutes, past
|
|
7729
|
+
* the 3 to 10 minutes a resident Worker rollover takes to settle) and a
|
|
7730
|
+
* timeout equal to the method's own budget (never above the engine's 30
|
|
7731
|
+
* minutes). Inputs to a step are the event's `resource`, the instance id and
|
|
7732
|
+
* previous steps' returns — refs, shas, a key, a path — never a payload and
|
|
7733
|
+
* never a credential. A step killed from outside (the container replaced
|
|
7734
|
+
* under it) throws and the engine retries it into the same idempotent
|
|
7735
|
+
* method; a gate that ends the cycle (idle, a container restart) or a
|
|
7736
|
+
* failure of the repository's own (recorded as `degraded`, the last snapshot
|
|
7737
|
+
* still serving) ends the instance with that word, and the next cron firing
|
|
7738
|
+
* creates the next one from the row's state. The instance runs one cycle
|
|
7739
|
+
* and returns: it is created by the watchdog cron per resident and
|
|
7740
|
+
* ten-minute bucket (`createRefreshInstance`), never a loop.
|
|
7741
|
+
*
|
|
7742
|
+
* The run is the cycle's root span, `resident.refresh` carrying the instance
|
|
7743
|
+
* id (docs/reference/specs/tracing.md item 25), with every command a step ran
|
|
7744
|
+
* grafted under it as a `resident.<step>` child — the same shape the alarm's
|
|
7745
|
+
* root has. */
|
|
7746
|
+
export class ResidentRefresh extends WorkflowEntrypoint<Env, RefreshInstanceParams> {
|
|
7747
|
+
async run(
|
|
7748
|
+
event: Readonly<WorkflowEvent<RefreshInstanceParams>>,
|
|
7749
|
+
step: WorkflowStep,
|
|
7750
|
+
): Promise<RefreshInstanceSummary> {
|
|
7751
|
+
const { resource } = event.payload;
|
|
7752
|
+
const instance = event.instanceId;
|
|
7753
|
+
const stub = residentStub(this.env, resource);
|
|
7754
|
+
const root = startAdoptedRoot(tracer, "resident.refresh", {
|
|
7755
|
+
sinks: traceSinks,
|
|
7756
|
+
startedAt: event.timestamp.getTime(),
|
|
7757
|
+
attrs: { instanceId: instance },
|
|
7758
|
+
});
|
|
7759
|
+
const graft = (answer: InstanceStepTrace) =>
|
|
7760
|
+
graftResidentSteps(answer.trace, {
|
|
7761
|
+
parent: root,
|
|
7762
|
+
prefix: "resident",
|
|
7763
|
+
baseAt: answer.startedAt,
|
|
7764
|
+
clipAt: systemClock(),
|
|
7765
|
+
});
|
|
7766
|
+
const retries = REFRESH_STEP_RETRIES;
|
|
7767
|
+
/** The word a step that did not finish ends the instance with, and its outcome for the root. */
|
|
7768
|
+
const ended = (
|
|
7769
|
+
at: RefreshInstanceSummary["step"],
|
|
7770
|
+
answer: { status: "stopped"; why: string } | { status: "failed"; reason: string },
|
|
7771
|
+
): RefreshInstanceSummary => ({
|
|
7772
|
+
instance,
|
|
7773
|
+
outcome: answer.status === "stopped" ? answer.why : "failed",
|
|
7774
|
+
step: at,
|
|
7775
|
+
});
|
|
7776
|
+
let summary: RefreshInstanceSummary | undefined;
|
|
7777
|
+
try {
|
|
7778
|
+
const fetched = await step.do("fetch", { retries, timeout: stepTimeoutMs(REFRESH_FETCH_STEP_BUDGET_MS) }, () =>
|
|
7779
|
+
stub.refreshInstanceFetch({ resource, instance }),
|
|
7780
|
+
);
|
|
7781
|
+
graft(fetched);
|
|
7782
|
+
if (fetched.status !== "done") return (summary = ended("fetch", fetched));
|
|
7783
|
+
let depsEntry: string | null = null;
|
|
7784
|
+
if (fetched.install) {
|
|
7785
|
+
const installed = await step.do(
|
|
7786
|
+
"install",
|
|
7787
|
+
{ retries, timeout: stepTimeoutMs(REFRESH_INSTALL_STEP_BUDGET_MS) },
|
|
7788
|
+
() => stub.refreshInstanceInstall({ resource, instance, sha: fetched.sha, lockfileKey: fetched.lockfileKey }),
|
|
7789
|
+
);
|
|
7790
|
+
graft(installed);
|
|
7791
|
+
if (installed.status !== "done") return (summary = ended("install", installed));
|
|
7792
|
+
depsEntry = installed.entry;
|
|
7793
|
+
}
|
|
7794
|
+
if (fetched.action !== "unchanged") {
|
|
7795
|
+
const built = await step.do("build", { retries, timeout: stepTimeoutMs(REFRESH_BUILD_STEP_BUDGET_MS) }, () =>
|
|
7796
|
+
stub.refreshInstanceBuild({
|
|
7797
|
+
resource,
|
|
7798
|
+
instance,
|
|
7799
|
+
sha: fetched.sha,
|
|
7800
|
+
factsSha: fetched.factsSha,
|
|
7801
|
+
lockfileKey: fetched.lockfileKey,
|
|
7802
|
+
depsEntry,
|
|
7803
|
+
}),
|
|
7804
|
+
);
|
|
7805
|
+
graft(built);
|
|
7806
|
+
if (built.status !== "done") return (summary = ended("build", built));
|
|
7807
|
+
}
|
|
7808
|
+
const snapped = await step.do(
|
|
7809
|
+
"snapshot",
|
|
7810
|
+
{ retries, timeout: stepTimeoutMs(REFRESH_SNAPSHOT_STEP_BUDGET_MS) },
|
|
7811
|
+
() =>
|
|
7812
|
+
stub.refreshInstanceSnapshot({
|
|
7813
|
+
resource,
|
|
7814
|
+
instance,
|
|
7815
|
+
ref: fetched.ref,
|
|
7816
|
+
sha: fetched.sha,
|
|
7817
|
+
lockfileKey: fetched.lockfileKey,
|
|
7818
|
+
action: fetched.action,
|
|
7819
|
+
mintError: fetched.mintError,
|
|
7820
|
+
}),
|
|
7821
|
+
);
|
|
7822
|
+
graft(snapped);
|
|
7823
|
+
if (snapped.status !== "done") return (summary = ended("snapshot", snapped));
|
|
7824
|
+
return (summary = { instance, outcome: "ok", step: "snapshot", action: fetched.action, sha: fetched.sha });
|
|
7825
|
+
} catch (err) {
|
|
7826
|
+
// A step out of retries: the engine records the failed instance by id;
|
|
7827
|
+
// the row keeps its last state and the next cron firing starts the next cycle.
|
|
7828
|
+
root.fail(err);
|
|
7829
|
+
throw err;
|
|
7830
|
+
} finally {
|
|
7831
|
+
const outcome = summary?.outcome ?? "error";
|
|
7832
|
+
root.end(outcome === "ok" || (summary !== undefined && outcome !== "failed") ? "ok" : "error", { outcome });
|
|
7833
|
+
}
|
|
7834
|
+
}
|
|
7835
|
+
}
|
|
7836
|
+
|
|
6630
7837
|
function json(data: unknown, status = 200): Response {
|
|
6631
7838
|
// Item 62: every non-streamed body leaves through here; its `error`,
|
|
6632
7839
|
// `reason` and `summary` strings are made safe at the exit.
|