@coreplane/switchboard 1.202.1 → 1.203.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/config/config.example.yaml +10 -0
- package/dist/assets/deploy/cloudflare-resident/preflight.mjs +3 -3
- package/dist/assets/deploy/cloudflare-resident/refresh.ts +179 -93
- package/dist/assets/deploy/cloudflare-resident/shared.ts +15 -10
- package/dist/assets/deploy/cloudflare-resident/worker.ts +459 -632
- package/dist/assets/package-lock.json +3 -3
- package/dist/assets/package.json +1 -1
- package/dist/assets/source.json +3 -3
- package/dist/assets/src/core/authz/policy.ts +4 -0
- package/dist/assets/src/core/runRecord.ts +11 -0
- package/dist/assets/src/core/schedules.ts +12 -6
- package/dist/assets/src/core/ship/handoff.ts +246 -0
- package/dist/assets/src/execution/residentDisk.ts +2 -2
- package/dist/assets/src/execution/residentInstanceId.ts +48 -45
- package/dist/assets/src/execution/residentRefresh.ts +17 -66
- package/dist/assets/src/execution/residentState.ts +8 -9
- package/dist/assets/src/execution/residentStepPlan.ts +1 -1
- package/dist/assets/web/dist/.vite/manifest.json +55 -44
- package/dist/assets/web/dist/assets/{AppShell-CJLPO_aI.js → AppShell-Bw3-_TTE.js} +1 -1
- package/dist/assets/web/dist/assets/{CostsPage-C6ZvnAmH.js → CostsPage-XQ-fQzdz.js} +1 -1
- package/dist/assets/web/dist/assets/DeliveryPage-BRwcQyr7.js +1 -0
- package/dist/assets/web/dist/assets/{NotFoundPage-BdLcY60r.js → NotFoundPage-RjH-9dPy.js} +1 -1
- package/dist/assets/web/dist/assets/{ResidentDetailPage-hXWrr8T6.js → ResidentDetailPage-BRy5wkv9.js} +1 -1
- package/dist/assets/web/dist/assets/{ResidentsIndexPage-YLu4JwxJ.js → ResidentsIndexPage-DAEVLN8j.js} +1 -1
- package/dist/assets/web/dist/assets/{RunRoutePage-BDhHRJVO.js → RunRoutePage-B1KHmkZ9.js} +1 -1
- package/dist/assets/web/dist/assets/{RunsIndexPage-D7zZ5a0l.js → RunsIndexPage-3hUWFpFV.js} +1 -1
- package/dist/assets/web/dist/assets/{RunsTabs-BOUlSa2W.js → RunsTabs-Qn_5TVJU.js} +1 -1
- package/dist/assets/web/dist/assets/{ScheduledPage-BciOQcC-.js → ScheduledPage-DysVvVm1.js} +1 -1
- package/dist/assets/web/dist/assets/{StatusDot-CTDLBv92.js → StatusDot-Bug2a6T6.js} +1 -1
- package/dist/assets/web/dist/assets/{Tooltip-DD2v9Gxx.js → Tooltip-C2eEUbwn.js} +1 -1
- package/dist/assets/web/dist/assets/{favicon-C4Q8cRsP.js → favicon-CFfbl2AI.js} +1 -1
- package/dist/assets/web/dist/assets/{main-DrSlUcMg.js → main-0uhp-taL.js} +3 -3
- package/dist/assets/web/dist/assets/main-Xickfv8V.css +1 -0
- package/dist/cli.js +2932 -1875
- package/package.json +1 -1
- package/dist/assets/web/dist/assets/main-DoTjwE-G.css +0 -1
|
@@ -17,12 +17,15 @@
|
|
|
17
17
|
// operator scope POST /attach /detach /exec /read /write /op GET /status (state, reason, inFlight)
|
|
18
18
|
// unauthenticated GET /healthz (deploy wake ping; touches no DO)
|
|
19
19
|
//
|
|
20
|
-
// Lifecycle engine:
|
|
21
|
-
// stamped snapshot → warm), wake-path rehydration
|
|
22
|
-
// BEFORE restore, stamped snapshots refused on
|
|
23
|
-
// refresh
|
|
24
|
-
//
|
|
25
|
-
//
|
|
20
|
+
// Lifecycle engine: schedule-driven provisioning (clone → install/build →
|
|
21
|
+
// stamped snapshot → warm; the one timer left), wake-path rehydration
|
|
22
|
+
// (`restoring` persisted BEFORE restore, stamped snapshots refused on
|
|
23
|
+
// mismatch), the refresh cycle as a Workflow instance the cron creates per
|
|
24
|
+
// resident and ten-minute bucket (refresh.ts: fetch, install, build,
|
|
25
|
+
// snapshot, then the worktree sweep and the disk measurement as steps), and
|
|
26
|
+
// a cron watchdog (create the due instances; name a stale mid-flight marker
|
|
27
|
+
// by its instance; time out stuck onboarding; auto-rebuild after N
|
|
28
|
+
// consecutive down passes on a rehydration-flavored reason). GitHub App tokens are minted
|
|
26
29
|
// repo-scoped on WebCrypto. POST /rebuild is the down→onboarding escape hatch
|
|
27
30
|
// (discard snapshots, reprovision from scratch); /offboard and /rebuild
|
|
28
31
|
// support dryRun (itemized plan, nothing executed); onboard verifies GitHub
|
|
@@ -131,7 +134,6 @@ import {
|
|
|
131
134
|
classifyRefreshFailure,
|
|
132
135
|
restoreFailureDisposition,
|
|
133
136
|
killStaleBuildProcessesCommand,
|
|
134
|
-
nextRefreshDelayS,
|
|
135
137
|
planRefresh,
|
|
136
138
|
RUNTIME_REPLACEMENT_WORDING,
|
|
137
139
|
judgeRestoreProgress,
|
|
@@ -144,7 +146,6 @@ import {
|
|
|
144
146
|
type RefreshPlan,
|
|
145
147
|
type RestoreSample,
|
|
146
148
|
type RefreshFailure,
|
|
147
|
-
type RefreshOutcome,
|
|
148
149
|
} from "../../src/execution/residentRefresh.js";
|
|
149
150
|
import {
|
|
150
151
|
lifecycleOf,
|
|
@@ -250,13 +251,13 @@ import {
|
|
|
250
251
|
planDepsMaterialization,
|
|
251
252
|
} from "../../src/execution/residentDepsStore.js";
|
|
252
253
|
import { buildId, injectedBuildStamp } from "../../src/deploy/buildStamp.js";
|
|
253
|
-
import { createRefreshInstance, type RefreshInstanceParams } from "./refresh";
|
|
254
|
+
import { createRefreshInstance, createRefreshInstanceNow, type RefreshInstanceParams } from "./refresh";
|
|
254
255
|
import {
|
|
255
256
|
DEFAULT_EXEC_TIMEOUT_MS,
|
|
256
257
|
DEPS_STEP_OVERHEAD_MS,
|
|
257
258
|
errMsg,
|
|
258
259
|
GIT_NETWORK_TIMEOUT_MS,
|
|
259
|
-
|
|
260
|
+
THREAD_POOL_SIZE,
|
|
260
261
|
R2_TRANSFER_TIMEOUT_MS,
|
|
261
262
|
REFRESH_BUILD_TIMEOUT_MS,
|
|
262
263
|
REFRESH_INSTALL_TIMEOUT_MS,
|
|
@@ -317,8 +318,7 @@ export interface Env {
|
|
|
317
318
|
BACKUP_BUCKET: R2Bucket;
|
|
318
319
|
/** The refresh cycle as a Workflow instance (docs/reference/specs/resident-repos.md
|
|
319
320
|
* item 7): `ResidentRefresh` in refresh.ts, re-exported above. The watchdog
|
|
320
|
-
* cron creates one per resident
|
|
321
|
-
* `alarm` resident (the default) never has one. */
|
|
321
|
+
* cron creates one per resident and ten-minute bucket. */
|
|
322
322
|
RESIDENT_REFRESH: Workflow<RefreshInstanceParams>;
|
|
323
323
|
// Presigned snapshot transfers (docs/reference/specs/resident-repos.md item 61): with all
|
|
324
324
|
// four present the container moves archive bytes itself over presigned R2
|
|
@@ -436,7 +436,7 @@ const OPS_DIR = "/workspace/ops";
|
|
|
436
436
|
* repo are genuinely concurrent. Memory, not this list, is the real ceiling
|
|
437
437
|
* — see the instance_type note in wrangler.jsonc. Must match the useradd loop
|
|
438
438
|
* in the Dockerfile. */
|
|
439
|
-
const THREAD_USERS = Array.from({ length:
|
|
439
|
+
const THREAD_USERS = Array.from({ length: THREAD_POOL_SIZE }, (_, i) => `worker${i + 2}`);
|
|
440
440
|
|
|
441
441
|
/** Force-detach: after killing the thread user's processes, how long
|
|
442
442
|
* to wait for the in-flight op counter to drain (polled every
|
|
@@ -457,21 +457,16 @@ const FORCE_DETACH_KILL_TIMEOUT_MS = 2_000;
|
|
|
457
457
|
* binding record is KEPT so the next attach recreates with the same
|
|
458
458
|
* ref. Overridable per resident via the onboard-time `worktreeTtlDays`. */
|
|
459
459
|
const WORKTREE_TTL_DAYS_DEFAULT = 7;
|
|
460
|
-
/** The sweep self-reschedules hourly (armed by attach when no sweep pends). */
|
|
461
|
-
const SWEEP_INTERVAL_S = 60 * 60; // hourly: the sweep is the backstop for trees a run kept (dirty) or never released
|
|
462
460
|
/** A live binding whose last attach is older than this AND whose tree is clean
|
|
463
|
-
* (no uncommitted/unpushed work) is released by the
|
|
464
|
-
* ended before /detach existed, or whose
|
|
465
|
-
* keep to the TTL. */
|
|
461
|
+
* (no uncommitted/unpushed work) is released by the sweep step (every refresh
|
|
462
|
+
* instance runs one) — runs that ended before /detach existed, or whose
|
|
463
|
+
* release call was lost. Dirty trees keep to the TTL. */
|
|
466
464
|
const CLEAN_IDLE_RELEASE_S = 60 * 60;
|
|
467
|
-
/** Slack over SWEEP_INTERVAL_S before a pending sweep row counts as config
|
|
468
|
-
* drift (armed by older code with a longer interval). A healthy row is due at
|
|
469
|
-
* most SWEEP_INTERVAL_S out and only gets closer, so this never trips on one. */
|
|
470
|
-
const SWEEP_DRIFT_SLACK_S = 5 * 60;
|
|
471
465
|
/** Idle sleep: when no thread has attached within this window and no live
|
|
472
|
-
* tree is dirty, the refresh
|
|
473
|
-
*
|
|
474
|
-
* first if the mirror is stale
|
|
466
|
+
* tree is dirty, the refresh cycle skips the fetch and the cron holds the
|
|
467
|
+
* next instance to the idle cadence, so the container can actually sleep
|
|
468
|
+
* (SLEEP_AFTER); the next attach refreshes first if the mirror is stale
|
|
469
|
+
* (refresh-on-attach). */
|
|
475
470
|
const IDLE_AFTER_S = 60 * 60;
|
|
476
471
|
/** LRU eviction floor: an over-cap onboard with `evictColdest:true` may
|
|
477
472
|
* offboard the coldest eligible warm resident, but never one whose last
|
|
@@ -487,49 +482,42 @@ const GITHUB_API_TIMEOUT_MS = 10_000;
|
|
|
487
482
|
* idle-park like a warm one; the next attach still refreshes first. */
|
|
488
483
|
const DEGRADED_PARK_AFTER_CYCLES = 3;
|
|
489
484
|
const DEGRADED_STREAK_KEY = "resident:degradedStreak";
|
|
490
|
-
/** Degraded reasons
|
|
491
|
-
*
|
|
492
|
-
*
|
|
493
|
-
*
|
|
494
|
-
*
|
|
495
|
-
*
|
|
496
|
-
*
|
|
497
|
-
*
|
|
498
|
-
*
|
|
499
|
-
*
|
|
500
|
-
|
|
501
|
-
*
|
|
502
|
-
*
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
const INTERRUPTED_STREAK_KEY = "resident:interruptedStreak";
|
|
485
|
+
/** Degraded reasons that are not evidence about the repository — both always
|
|
486
|
+
* carry a `: detail` suffix: the watchdog's `stale-mid-flight: …` (a marker
|
|
487
|
+
* a dead cycle left behind; a cycle must run) and the wake path's
|
|
488
|
+
* `restore-interrupted: …` (the runtime was replaced under a restore; the
|
|
489
|
+
* step's retry restores again). They never count toward the park streak.
|
|
490
|
+
* Deliberate trade-off: a resident that oscillates between a
|
|
491
|
+
* refresh-produced failure and a watchdog stamp (e.g. `install-failed` → DO
|
|
492
|
+
* eviction → `stale-mid-flight` → `install-failed` …) keeps resetting the
|
|
493
|
+
* streak and never parks — full 10-min cadence for a chronically broken
|
|
494
|
+
* repo. Accepted: a watchdog stamp means the previous "same reason"
|
|
495
|
+
* observation is not trustworthy, and preserving the streak across it would
|
|
496
|
+
* re-open the parked-degraded hole this fixes. A refresh step killed from
|
|
497
|
+
* outside (`refresh-interrupted`, `classifyRefreshFailure`) is never recorded
|
|
498
|
+
* as `degraded` at all: the instance throws it to the engine, whose retry
|
|
499
|
+
* re-enters the step. */
|
|
500
|
+
const NON_EVIDENCE_REASON = /^(?:stale-mid-flight|restore-interrupted):/;
|
|
507
501
|
/** When the disk-full recovery last stopped the container (docs/reference/specs/resident-repos.md item 54):
|
|
508
502
|
* feeds `planDiskFullRecovery`'s cooldown so a working set that refills the
|
|
509
503
|
* disk is named, not recycled in a loop. */
|
|
510
504
|
const DISK_FULL_RECYCLE_KEY = "resident:diskFullRecycleAt";
|
|
511
|
-
/** A disk-full attach pulls the refresh cycle this close (seconds) so the
|
|
512
|
-
* recovery decision runs now, not at the next 600 s alarm. */
|
|
513
|
-
const DISK_FULL_REARM_S = 1;
|
|
514
505
|
/** The last disk measurement (docs/reference/specs/resident-repos.md item 55; `residentDiskBudget.ts`): one
|
|
515
|
-
* `df` + one `du` over the parts, taken
|
|
516
|
-
*
|
|
517
|
-
*
|
|
518
|
-
*
|
|
506
|
+
* `df` + one `du` over the parts, taken by every refresh instance's `measure`
|
|
507
|
+
* step after its sweep. Surfaced as the live view's `disk`; the attach
|
|
508
|
+
* admission projects a new tree's cost from its parts (its free-space term
|
|
509
|
+
* is a live `df` of its own). */
|
|
519
510
|
const DISK_KEY = "resident:disk";
|
|
520
|
-
/**
|
|
521
|
-
*
|
|
522
|
-
*
|
|
523
|
-
*
|
|
524
|
-
*
|
|
525
|
-
* both drive a cycle for one resident. */
|
|
511
|
+
/** The lifecycle row (docs/reference/specs/resident-repos.md item 7): `workflow`,
|
|
512
|
+
* the one scheduler. Kept from the flagged rollout so `/status` and `/debug
|
|
513
|
+
* info` can say so; an `alarm` value a flip left behind reads `workflow`
|
|
514
|
+
* (`lifecycleOf`) — the chain it named no longer exists. Rewritten by the
|
|
515
|
+
* admin `/debug` `lifecycle` op. */
|
|
526
516
|
const LIFECYCLE_KEY = "resident:lifecycle";
|
|
527
|
-
/** The refresh instance row (item 7): the last instance
|
|
528
|
-
*
|
|
529
|
-
*
|
|
517
|
+
/** The refresh instance row (item 7): the last instance created for this
|
|
518
|
+
* resident, with the step it last reported and the cycle lease it holds, and
|
|
519
|
+
* the last bucket the cron skipped (a live cycle, a duplicate id). */
|
|
530
520
|
const REFRESH_INSTANCE_KEY = "resident:refreshInstance";
|
|
531
|
-
const DISK_MEASURE_CALLBACK = "onDiskMeasure";
|
|
532
|
-
const DISK_MEASURE_DELAY_S = 1;
|
|
533
521
|
/** A `du` over a multi-GB checkout plus every live tree is seconds warm, tens
|
|
534
522
|
* of seconds on a cold page cache — the same class as a git network step. */
|
|
535
523
|
const DU_TIMEOUT_MS = GIT_NETWORK_TIMEOUT_MS;
|
|
@@ -1030,7 +1018,7 @@ interface RepoFacts {
|
|
|
1030
1018
|
lastRefreshAt: string;
|
|
1031
1019
|
lastRefreshError?: string; // last cycle's failure reason: command-level (e.g. token mint, no lifecycle flip) or the classified reason of a failed/interrupted cycle (survives a concurrent state overwrite); cleared by the next completed cycle
|
|
1032
1020
|
lastRestore?: { at: string; ms: number }; // proof of restore-not-reclone on the wake path
|
|
1033
|
-
/** Set while the resident is in idle mode (refresh
|
|
1021
|
+
/** Set while the resident is in idle mode (the cron holds the next refresh instance to the idle cadence so the container may sleep). */
|
|
1034
1022
|
idleSince?: string;
|
|
1035
1023
|
}
|
|
1036
1024
|
|
|
@@ -1083,6 +1071,17 @@ interface RefreshInstanceRow {
|
|
|
1083
1071
|
skipped: { id: string; at: string; why: string } | null;
|
|
1084
1072
|
}
|
|
1085
1073
|
|
|
1074
|
+
/** The engine's statuses under which an instance is still a live cycle:
|
|
1075
|
+
* queued, running, paused or waiting. Anything else — `complete`, `errored`,
|
|
1076
|
+
* `terminated`, an id the engine does not know — is not. */
|
|
1077
|
+
const INSTANCE_LIVE_STATUSES: ReadonlySet<string> = new Set([
|
|
1078
|
+
"queued",
|
|
1079
|
+
"running",
|
|
1080
|
+
"paused",
|
|
1081
|
+
"waiting",
|
|
1082
|
+
"waitingForPause",
|
|
1083
|
+
]);
|
|
1084
|
+
|
|
1086
1085
|
/** What every instance step answers besides its own facts: the resident's
|
|
1087
1086
|
* wall clock at the step's start and the commands it ran, so the instance
|
|
1088
1087
|
* can graft them under its root the way the bot grafts an attach's. */
|
|
@@ -1242,8 +1241,16 @@ export class ResidentRegistryDO extends DurableObject<Env> {
|
|
|
1242
1241
|
|
|
1243
1242
|
const PROVISIONING_CALLBACK = "onProvisioningDeadline"; // fail-closed deadline
|
|
1244
1243
|
const PROVISION_RUN_CALLBACK = "runProvisioning"; // the actual provisioning work
|
|
1245
|
-
|
|
1246
|
-
|
|
1244
|
+
/** The schedule callbacks the retired alarm chain armed, gone with the flip to
|
|
1245
|
+
* the Workflow scheduler. Their rows outlive the code that armed them, and the
|
|
1246
|
+
* SDK's `alarm()` skips a due row whose callback method is gone WITHOUT
|
|
1247
|
+
* deleting it, then re-arms for that past time at once — a hot alarm loop
|
|
1248
|
+
* that only the row's deletion ends (@cloudflare/containers 0.3.7,
|
|
1249
|
+
* dist/lib/container.js `alarm()`: the `continue` precedes the DELETE). The
|
|
1250
|
+
* constructor deletes them before the first event, which on an upgraded
|
|
1251
|
+
* resident is that very alarm. Drop this list once every resident has woken on
|
|
1252
|
+
* this code (lifecycle.test.ts pins the names). */
|
|
1253
|
+
const RETIRED_SCHEDULE_CALLBACKS = ["onRefreshAlarm", "onWorktreeSweep", "onDiskMeasure"] as const;
|
|
1247
1254
|
|
|
1248
1255
|
const STATE_KEY = "resident:state";
|
|
1249
1256
|
const REASON_KEY = "resident:reason";
|
|
@@ -1298,20 +1305,44 @@ class StepError extends Error {
|
|
|
1298
1305
|
}
|
|
1299
1306
|
|
|
1300
1307
|
/** Thrown after the resident has ALREADY been transitioned to `down` (reason
|
|
1301
|
-
* persisted); signals callers
|
|
1308
|
+
* persisted); signals callers the cycle is over without re-flipping. */
|
|
1302
1309
|
class ResidentDownError extends Error {
|
|
1303
1310
|
constructor(public reason: string) {
|
|
1304
1311
|
super(reason);
|
|
1305
1312
|
}
|
|
1306
1313
|
}
|
|
1307
1314
|
|
|
1315
|
+
/** Thrown by a refresh gate that stopped the container on purpose — a stale
|
|
1316
|
+
* image (`reconcileImage`), a disk-full recycle (`recoverFromDiskFull`) — so
|
|
1317
|
+
* the instance step it runs in throws to the engine, whose retry (thirty
|
|
1318
|
+
* seconds on) finds the container back on the current image or an empty
|
|
1319
|
+
* disk, restores it and runs the cycle: the retry is the re-warm that a
|
|
1320
|
+
* short re-arm used to be. Not a failure of the repository's own — never
|
|
1321
|
+
* recorded as `degraded`. */
|
|
1322
|
+
class CycleRestartError extends Error {
|
|
1323
|
+
constructor(public why: "image-stale-restart" | "disk-full-restart") {
|
|
1324
|
+
super(`${why}: the container is restarting — the step is retried onto it`);
|
|
1325
|
+
}
|
|
1326
|
+
}
|
|
1327
|
+
|
|
1308
1328
|
export class ResidentDO extends Sandbox<Env> {
|
|
1309
|
-
// TIMER RULE: never
|
|
1310
|
-
//
|
|
1311
|
-
//
|
|
1312
|
-
//
|
|
1313
|
-
//
|
|
1314
|
-
//
|
|
1329
|
+
// TIMER RULE: lifecycle code never arms the Durable Object's own alarm slot
|
|
1330
|
+
// — the Container base class owns it (its sleepAfter machinery and the
|
|
1331
|
+
// schedule multiplexing live there). The one timer left is provisioning's
|
|
1332
|
+
// (`initResident`: the run at +1 s and its fail-closed deadline), through
|
|
1333
|
+
// the base class's schedule API, which multiplexes onto that slot safely
|
|
1334
|
+
// (checked against @cloudflare/containers 0.3.7: the SDK registers no
|
|
1335
|
+
// schedule callback names, so ours cannot collide). Every other cycle is a
|
|
1336
|
+
// Workflow instance (refresh.ts) and the watchdog re-arms nothing;
|
|
1337
|
+
// lifecycle.test.ts holds the line over these sources.
|
|
1338
|
+
|
|
1339
|
+
constructor(...args: ConstructorParameters<typeof Sandbox<Env>>) {
|
|
1340
|
+
super(...args);
|
|
1341
|
+
// The base class created `container_schedules` synchronously above; the
|
|
1342
|
+
// rows the retired alarm chain armed are gone before this object handles
|
|
1343
|
+
// its first event (RETIRED_SCHEDULE_CALLBACKS).
|
|
1344
|
+
for (const name of RETIRED_SCHEDULE_CALLBACKS) this.deleteSchedules(name);
|
|
1345
|
+
}
|
|
1315
1346
|
|
|
1316
1347
|
/** Serializes concurrent hydration attempts within one DO lifetime. Never
|
|
1317
1348
|
* used as a "hydrated" flag — the container can sleep while the DO object
|
|
@@ -1978,8 +2009,8 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
1978
2009
|
// about to change and asks the pure plan (residentStepPlan.ts) whether the
|
|
1979
2010
|
// work is done — done issues no command, so a second call with the same
|
|
1980
2011
|
// inputs has no effect — then takes its lease, runs its commands under the
|
|
1981
|
-
// step's own budget, writes its result and releases. The
|
|
1982
|
-
// them
|
|
2012
|
+
// step's own budget, writes its result and releases. The refresh instance
|
|
2013
|
+
// drives them one step at a time (`refreshInstance*`).
|
|
1983
2014
|
|
|
1984
2015
|
/** Fetch the mirror from origin, once per cycle: the record under
|
|
1985
2016
|
* LAST_FETCH_KEY names the cycle, so a repeated call inside the same cycle
|
|
@@ -2205,9 +2236,9 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2205
2236
|
* or capped restore leaves the resident `down` with the reason and the
|
|
2206
2237
|
* container stopped, as the wake path always did; a restore the runtime
|
|
2207
2238
|
* replacement interrupts (a deploy rolled the container under it) is
|
|
2208
|
-
* `restore-interrupted`, degraded and rethrown for the
|
|
2209
|
-
*
|
|
2210
|
-
* (restoreFailureDisposition). */
|
|
2239
|
+
* `restore-interrupted`, degraded and rethrown for the instance step that
|
|
2240
|
+
* called it to retry — nothing is streaming into a disk that no longer
|
|
2241
|
+
* exists (restoreFailureDisposition). */
|
|
2211
2242
|
async restoreCheckout(snap: SnapshotRecord, deadlineMs: number): Promise<{ done: boolean }> {
|
|
2212
2243
|
const plan = planRestore({ sha: snap.sha, readyStamp: await this.readyStamp() });
|
|
2213
2244
|
if (plan.action === "done") return { done: true };
|
|
@@ -2265,9 +2296,9 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2265
2296
|
// container, so nothing can land on a rebuild and there is nothing to
|
|
2266
2297
|
// stop. Not evidence about the repo — the resident is `degraded` with
|
|
2267
2298
|
// the restore named, never `down`, and the error goes back to the
|
|
2268
|
-
//
|
|
2269
|
-
// and
|
|
2270
|
-
// container. Before this branch every such restore ended `down`, and
|
|
2299
|
+
// instance step, whose classifier reads the same wording as an
|
|
2300
|
+
// interruption and throws to the engine; the retry restores again onto
|
|
2301
|
+
// the new container. Before this branch every such restore ended `down`, and
|
|
2271
2302
|
// only a rebuild (the watchdog's, after three passes) brought the
|
|
2272
2303
|
// resident back.
|
|
2273
2304
|
this.swapIncarnation(); // the container this incarnation's memos described is gone
|
|
@@ -2325,16 +2356,11 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2325
2356
|
return counts.reduce((a, n) => a + n, 0);
|
|
2326
2357
|
}
|
|
2327
2358
|
|
|
2328
|
-
|
|
2329
|
-
|
|
2330
|
-
|
|
2331
|
-
}
|
|
2332
|
-
|
|
2333
|
-
/** Persist `down` with a reason, stop the refresh chain, and hand back the
|
|
2334
|
-
* error that tells callers the transition already happened. */
|
|
2359
|
+
/** Persist `down` with a reason and hand back the error that tells callers
|
|
2360
|
+
* the transition already happened (a down resident gets no refresh
|
|
2361
|
+
* instance: the cron's decision reads the state). */
|
|
2335
2362
|
private async goDown(reason: string): Promise<ResidentDownError> {
|
|
2336
2363
|
await this.setResidentState("down", reason);
|
|
2337
|
-
this.deleteSchedules(REFRESH_CALLBACK);
|
|
2338
2364
|
return new ResidentDownError(reason);
|
|
2339
2365
|
}
|
|
2340
2366
|
|
|
@@ -2360,13 +2386,12 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2360
2386
|
await this.ctx.storage.delete([FACTS_KEY, SNAPSHOT_KEY]); // defensive: no stale facts from a past life
|
|
2361
2387
|
this.deleteSchedules(PROVISIONING_CALLBACK);
|
|
2362
2388
|
this.deleteSchedules(PROVISION_RUN_CALLBACK);
|
|
2363
|
-
this.deleteSchedules(REFRESH_CALLBACK);
|
|
2364
2389
|
await this.schedule(Math.max(1, Math.ceil(provisioningTimeoutMs / 1000)), PROVISIONING_CALLBACK, resource);
|
|
2365
2390
|
await this.schedule(1, PROVISION_RUN_CALLBACK, resource);
|
|
2366
2391
|
return { state: "onboarding", reason: "" };
|
|
2367
2392
|
}
|
|
2368
2393
|
|
|
2369
|
-
/** The provisioning engine (
|
|
2394
|
+
/** The provisioning engine (schedule-driven): clone bare mirror → resolve the
|
|
2370
2395
|
* default branch → full install + build in a working checkout using the
|
|
2371
2396
|
* onboard-time command table → stamped snapshot → record facts → warm.
|
|
2372
2397
|
* On failure: down(provision-failed at <step>) — the registry slot is
|
|
@@ -2481,7 +2506,8 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2481
2506
|
await this.writeDiskMarkers({ ready: sha, depsKey: lockfileHash, builtSha: sha });
|
|
2482
2507
|
this.deleteSchedules(PROVISIONING_CALLBACK);
|
|
2483
2508
|
await this.setResidentState("warm");
|
|
2484
|
-
|
|
2509
|
+
// The first refresh instance is the cron's: the row now reads `warm`
|
|
2510
|
+
// with no instance recorded, so the next firing creates it (item 9).
|
|
2485
2511
|
} catch (err) {
|
|
2486
2512
|
if ((await this.ctx.storage.get<ResidentState>(STATE_KEY)) !== "onboarding") return;
|
|
2487
2513
|
this.deleteSchedules(PROVISIONING_CALLBACK);
|
|
@@ -2507,7 +2533,6 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2507
2533
|
await this.setResidentState("down", reason);
|
|
2508
2534
|
this.deleteSchedules(PROVISIONING_CALLBACK);
|
|
2509
2535
|
this.deleteSchedules(PROVISION_RUN_CALLBACK);
|
|
2510
|
-
this.deleteSchedules(REFRESH_CALLBACK);
|
|
2511
2536
|
const resource = await this.ctx.storage.get<string>(RESOURCE_KEY);
|
|
2512
2537
|
if (resource) {
|
|
2513
2538
|
try {
|
|
@@ -2525,12 +2550,13 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2525
2550
|
* otherwise still say warm while the R2 restore runs. Refuses mismatched
|
|
2526
2551
|
* stamps → down(snapshot-stamp-mismatch); a stalled or capped restore →
|
|
2527
2552
|
* down(r2-restore-failed); a restore the runtime replacement interrupts →
|
|
2528
|
-
* degraded(restore-interrupted), rethrown so the
|
|
2529
|
-
*
|
|
2530
|
-
* refresh
|
|
2553
|
+
* degraded(restore-interrupted), rethrown so the instance step that called
|
|
2554
|
+
* it is retried by the engine. Throws ResidentDownError after the down
|
|
2555
|
+
* transitions. Called by the refresh instance's fetch step (and the attach
|
|
2556
|
+
* path). */
|
|
2531
2557
|
async ensureHydrated(): Promise<void> {
|
|
2532
2558
|
// Fresh positive verdict for this incarnation → nothing to probe. See the
|
|
2533
|
-
// per-incarnation memo block for why this is safe; the refresh
|
|
2559
|
+
// per-incarnation memo block for why this is safe; the refresh cycle's
|
|
2534
2560
|
// 10-min cadence always outlives the TTL, so a cycle re-probes for real.
|
|
2535
2561
|
if (this.hydratedVerdictAt !== 0 && systemClock() - this.hydratedVerdictAt < this.hydrationMemoTtlMs) return;
|
|
2536
2562
|
if (this.hydration) return this.hydration;
|
|
@@ -2671,164 +2697,27 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2671
2697
|
await this.setResidentState("warm");
|
|
2672
2698
|
}
|
|
2673
2699
|
|
|
2674
|
-
// -- freshness (the
|
|
2675
|
-
|
|
2676
|
-
/** Self-rescheduling refresh: rehydrate if the container slept → mint a
|
|
2677
|
-
* repo-scoped token (mint failure is command-level: recorded, never a
|
|
2678
|
-
* lifecycle flip) → fetch into the bare mirror → when the default branch
|
|
2679
|
-
* moved: plan against the disk checkpoints (planRefresh) — reuse a
|
|
2680
|
-
* checkout an interrupted cycle already materialized, else update the
|
|
2681
|
-
* checkout and reinstall ONLY if the committed lockfile key changed, then
|
|
2682
|
-
* rebuild — write a new stamped snapshot, delete the replaced backup
|
|
2683
|
-
* objects. Transitions: refreshing → warm, or degraded(reason) with the
|
|
2684
|
-
* last snapshot still serving. */
|
|
2685
|
-
async onRefreshAlarm(payload: string): Promise<void> {
|
|
2686
|
-
// The freshness cycle nobody asked for is a root of its own
|
|
2687
|
-
// (docs/reference/specs/tracing.md item 25): `resident.refresh`, with every command it
|
|
2688
|
-
// ran as a `resident.<step>` child, exactly like an attach's.
|
|
2689
|
-
const t0 = systemClock();
|
|
2690
|
-
const trace = createStepTrace(t0);
|
|
2691
|
-
let outcome = "ok";
|
|
2692
|
-
try {
|
|
2693
|
-
await this.stepTrace.run(trace, () => this.onRefreshAlarmTraced(payload));
|
|
2694
|
-
} catch (err) {
|
|
2695
|
-
outcome = "error";
|
|
2696
|
-
throw err;
|
|
2697
|
-
} finally {
|
|
2698
|
-
emitStepRoot("resident.refresh", t0, trace.steps(), undefined, outcome);
|
|
2699
|
-
}
|
|
2700
|
-
}
|
|
2701
|
-
|
|
2702
|
-
private async onRefreshAlarmTraced(payload: string): Promise<void> {
|
|
2703
|
-
const resource = payload || ((await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "");
|
|
2704
|
-
// Item 7: a resident on the Workflow lifecycle has no chain. An alarm a
|
|
2705
|
-
// previous flip left armed — or an attach's +1 s pull, or a provisioning's
|
|
2706
|
-
// first arm — runs nothing and re-arms nothing, so the two schedulers never
|
|
2707
|
-
// both drive a cycle for one resident.
|
|
2708
|
-
if ((await this.getLifecycle()) === "workflow") {
|
|
2709
|
-
console.log(`refresh: lifecycle is workflow — the alarm chain runs no cycle for ${resource}`);
|
|
2710
|
-
return;
|
|
2711
|
-
}
|
|
2712
|
-
let refreshCounted = false;
|
|
2713
|
-
/** This cycle's lease holder in the in-flight row, once it is counted. */
|
|
2714
|
-
let cycleHolder: string | null = null;
|
|
2715
|
-
/** This firing's identity: what `fetchMirror` records so a repeated call inside the cycle is done. */
|
|
2716
|
-
const cycle = crypto.randomUUID();
|
|
2717
|
-
const before = await this.getStatus();
|
|
2718
|
-
// down chains stay down (a rebuild is the escape hatch); onboarding is
|
|
2719
|
-
// owned by provisioning, which arms the first refresh itself.
|
|
2720
|
-
if (before.state === "onboarding" || before.state === "down") return;
|
|
2721
|
-
try {
|
|
2722
|
-
const gate = await this.refreshGate(resource);
|
|
2723
|
-
if (!gate.go) return; // finally re-arms on the outcome the gate set
|
|
2724
|
-
const { record, facts } = gate;
|
|
2725
|
-
// From here the cycle mutates the mirror/checkout: count it as in flight
|
|
2726
|
-
// so an attach-path reconcileImage never stops the container under it,
|
|
2727
|
-
// and lease it in the in-flight row so the watchdog can tell this cycle
|
|
2728
|
-
// from a marker a dead one left behind (item 22).
|
|
2729
|
-
this.refreshesInFlight++;
|
|
2730
|
-
refreshCounted = true;
|
|
2731
|
-
cycleHolder = this.nextHolder();
|
|
2732
|
-
await this.recordInFlight("refresh", cycleHolder, REFRESH_CYCLE_LEASE_MS, "refresh");
|
|
2733
|
-
|
|
2734
|
-
const fetched = await this.refreshFetch(resource, facts, cycle, 1);
|
|
2735
|
-
if (!fetched.ok) return;
|
|
2736
|
-
const { sha, lockfileHash, mintError, token } = fetched;
|
|
2737
|
-
const plan = planRefresh({
|
|
2738
|
-
sha,
|
|
2739
|
-
factsSha: facts.sha,
|
|
2740
|
-
lockfileKey: lockfileHash,
|
|
2741
|
-
disk: await this.readRefreshDisk(),
|
|
2742
|
-
});
|
|
2743
|
-
// Whether this cycle's snapshot step committed (`superseded` means
|
|
2744
|
-
// another writer moved the record, whose facts then stand).
|
|
2745
|
-
let committed = true;
|
|
2746
|
-
if (plan.action !== "unchanged") {
|
|
2747
|
-
const t0 = systemClock();
|
|
2748
|
-
console.log(`refresh: ${facts.sha.slice(0, 8)} → ${sha.slice(0, 8)}: ${plan.action} (${plan.why})`);
|
|
2749
|
-
// Deps come from the store (item 59): a changed lockfile key is
|
|
2750
|
-
// materialized ONCE into `/workspace/deps/<key>` — OUTSIDE the mirror
|
|
2751
|
-
// lock, because the install runs in its own scratch clone and touches
|
|
2752
|
-
// no consumer's tree (the staging step) — and the checkout's
|
|
2753
|
-
// node_modules becomes a hardlink view of that entry (runBuild). An
|
|
2754
|
-
// attach that needs the same key joins this very install instead of
|
|
2755
|
-
// starting its own. Checkpoint: the deps marker comes off BEFORE the
|
|
2756
|
-
// install so an interruption mid-install can never read as completion.
|
|
2757
|
-
let depsEntry: string | null = null;
|
|
2758
|
-
if (plan.action === "rebuild") {
|
|
2759
|
-
await this.refreshClearMarkers(plan.install);
|
|
2760
|
-
if (plan.install) depsEntry = await this.refreshInstall(record, facts, sha, lockfileHash);
|
|
2761
|
-
}
|
|
2762
|
-
// `reuse`: the checkout already holds this sha with its deps and build
|
|
2763
|
-
// (an interrupted cycle got that far) — the build step finds it done and
|
|
2764
|
-
// only the snapshot, facts and stamp are missing; they move together.
|
|
2765
|
-
await this.runBuild({
|
|
2766
|
-
sha,
|
|
2767
|
-
factsSha: facts.sha,
|
|
2768
|
-
lockfileKey: lockfileHash,
|
|
2769
|
-
buildCmd: record.commands.build,
|
|
2770
|
-
depsEntry,
|
|
2771
|
-
});
|
|
2772
|
-
committed = await this.refreshSnapshot(resource, { ref: facts.defaultRef, sha, lockfileHash });
|
|
2773
|
-
console.log(`refresh: ${sha.slice(0, 8)} ${plan.action} done in ${systemClock() - t0}ms`);
|
|
2774
|
-
}
|
|
2775
|
-
await this.refreshComplete(resource, facts, { sha, lockfileHash, committed, mintError, token });
|
|
2776
|
-
} catch (err) {
|
|
2777
|
-
if (err instanceof ResidentDownError) return; // already down with reason; chain stops below
|
|
2778
|
-
const failure = await this.classifyCycleError(err);
|
|
2779
|
-
// Set BEFORE the writes on purpose: if either throws, the finally still
|
|
2780
|
-
// re-arms short — the safe direction for an interruption.
|
|
2781
|
-
if (failure.interrupted) this.rearmOutcome = "interrupted";
|
|
2782
|
-
await this.refreshFailed(failure, refreshCounted ? 1 : 0);
|
|
2783
|
-
} finally {
|
|
2784
|
-
if (refreshCounted) this.refreshesInFlight--;
|
|
2785
|
-
if (cycleHolder) await this.clearInFlight("refresh", cycleHolder);
|
|
2786
|
-
const state = await this.ctx.storage.get<ResidentState>(STATE_KEY);
|
|
2787
|
-
// Consecutive-interruption count: bounds the short re-arm so
|
|
2788
|
-
// a step whose output chronically carries the kill signature falls back to
|
|
2789
|
-
// the cadence after INTERRUPTED_REARM_MAX_CONSECUTIVE instead of hot-looping.
|
|
2790
|
-
let consecutiveInterrupted: number | undefined;
|
|
2791
|
-
if (this.rearmOutcome === "interrupted") {
|
|
2792
|
-
consecutiveInterrupted = ((await this.ctx.storage.get<number>(INTERRUPTED_STREAK_KEY)) ?? 0) + 1;
|
|
2793
|
-
await this.ctx.storage.put(INTERRUPTED_STREAK_KEY, consecutiveInterrupted);
|
|
2794
|
-
} else {
|
|
2795
|
-
await this.ctx.storage.delete(INTERRUPTED_STREAK_KEY);
|
|
2796
|
-
}
|
|
2797
|
-
const interval = nextRefreshDelayS({
|
|
2798
|
-
outcome: this.rearmOutcome,
|
|
2799
|
-
intervalS: REFRESH_INTERVAL_S,
|
|
2800
|
-
idleIntervalS: IDLE_REFRESH_INTERVAL_S,
|
|
2801
|
-
consecutiveInterrupted,
|
|
2802
|
-
});
|
|
2803
|
-
this.rearmOutcome = "normal";
|
|
2804
|
-
// A flip to `workflow` while this cycle ran: the chain ends here (item 7).
|
|
2805
|
-
const chained = (await this.getLifecycle()) === "alarm";
|
|
2806
|
-
if (chained && state && state !== "down" && state !== "onboarding") await this.armRefresh(resource, interval);
|
|
2807
|
-
}
|
|
2808
|
-
}
|
|
2809
|
-
|
|
2810
|
-
// -- the cycle's phases, shared by the alarm and the instance (item 7) -------
|
|
2700
|
+
// -- freshness (the refresh cycle's phases, one per instance step) -----------
|
|
2811
2701
|
//
|
|
2812
|
-
// The
|
|
2813
|
-
//
|
|
2814
|
-
//
|
|
2815
|
-
//
|
|
2816
|
-
|
|
2817
|
-
|
|
2818
|
-
|
|
2819
|
-
*
|
|
2820
|
-
*
|
|
2821
|
-
*
|
|
2822
|
-
*
|
|
2823
|
-
* the park streak and the idle gate (`idle`); and, for a
|
|
2824
|
-
* the end of idle mode.
|
|
2825
|
-
* instance resets it. */
|
|
2702
|
+
// The cycle runs as the Workflow instance in refresh.ts: the gates, the
|
|
2703
|
+
// fetch, the plan, the install, the build, the snapshot, the completion,
|
|
2704
|
+
// then the housekeeping steps. Each phase is one method here; the instance
|
|
2705
|
+
// calls them one step at a time (`refreshInstance*`, below).
|
|
2706
|
+
|
|
2707
|
+
/** The cycle's entry gates, in order: hydrate; the registry record (gone →
|
|
2708
|
+
* the resident was offboarded mid-flight, nothing to do); the image
|
|
2709
|
+
* reconcile (a stale image stops the container, which restarts on the
|
|
2710
|
+
* current one — thrown as `CycleRestartError` so the engine retries the
|
|
2711
|
+
* step onto it); the disk-full re-probe (a disk still full decides its own
|
|
2712
|
+
* recovery — a recycle is thrown the same way, a kept container stops the
|
|
2713
|
+
* cycle — item 54); the park streak and the idle gate (`idle`); and, for a
|
|
2714
|
+
* cycle that runs, the end of idle mode. */
|
|
2826
2715
|
private async refreshGate(
|
|
2827
2716
|
resource: string,
|
|
2828
2717
|
): Promise<{ go: false; why: string } | { go: true; record: ResidentRecord; facts: RepoFacts }> {
|
|
2829
2718
|
await this.ensureHydrated();
|
|
2830
2719
|
const record = await this.registry().getRecord(resource);
|
|
2831
|
-
if (!record) return { go: false, why: "offboarded" }; // offboarded mid-flight:
|
|
2720
|
+
if (!record) return { go: false, why: "offboarded" }; // offboarded mid-flight: the cycle ends quietly
|
|
2832
2721
|
const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
|
|
2833
2722
|
if (!facts) throw new StepError("facts", "no repo facts recorded despite hydration");
|
|
2834
2723
|
|
|
@@ -2837,16 +2726,16 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2837
2726
|
// users the image lacks. Reconcile here (every cycle, cheap) — see
|
|
2838
2727
|
// reconcileImage — so a rollout self-applies within one refresh.
|
|
2839
2728
|
if (await this.reconcileImage("refresh")) {
|
|
2840
|
-
// Container stopping; it restarts on the new image in seconds.
|
|
2841
|
-
//
|
|
2842
|
-
//
|
|
2843
|
-
|
|
2844
|
-
return { go: false, why: "image-stale-restart" };
|
|
2729
|
+
// Container stopping; it restarts on the new image in seconds. The
|
|
2730
|
+
// engine's retry re-enters this step thirty seconds on and re-warms the
|
|
2731
|
+
// resident within the minute, instead of the next bucket.
|
|
2732
|
+
throw new CycleRestartError("image-stale-restart");
|
|
2845
2733
|
}
|
|
2846
2734
|
|
|
2847
2735
|
// Idle sleep: nobody has attached for IDLE_AFTER_S and no live tree is
|
|
2848
|
-
// dirty → skip this fetch
|
|
2849
|
-
// elapse. Staleness is repaid at the next
|
|
2736
|
+
// dirty → skip this fetch; the cron holds the next instance to the idle
|
|
2737
|
+
// cadence, so SLEEP_AFTER can elapse. Staleness is repaid at the next
|
|
2738
|
+
// attach (refreshIfStale). A
|
|
2850
2739
|
// dirty live tree pins the container awake: sleep destroys the disk and
|
|
2851
2740
|
// uncommitted work is not snapshotted.
|
|
2852
2741
|
// Only a SETTLED resident may park: a cycle that finds `refreshing`/
|
|
@@ -2868,8 +2757,11 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2868
2757
|
// sweep freed trees) → run the cycle as usual and earn `warm`.
|
|
2869
2758
|
const free = await this.freeKiB();
|
|
2870
2759
|
if (free !== null && free < DISK_FULL_FREE_KIB) {
|
|
2871
|
-
|
|
2872
|
-
|
|
2760
|
+
// A recycle: the engine's retry re-enters this step, restores onto the
|
|
2761
|
+
// empty disk and runs the cycle. A kept container: nothing a fetch can
|
|
2762
|
+
// do until space is freed; the cycle ends and the next bucket re-probes.
|
|
2763
|
+
if (await this.recoverFromDiskFull(entry.reason, 0)) throw new CycleRestartError("disk-full-restart");
|
|
2764
|
+
return { go: false, why: "disk-full" };
|
|
2873
2765
|
}
|
|
2874
2766
|
}
|
|
2875
2767
|
let settled = entry.state === "warm";
|
|
@@ -2886,11 +2778,10 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2886
2778
|
await this.ctx.storage.put(DEGRADED_STREAK_KEY, streak);
|
|
2887
2779
|
settled = streak.count >= DEGRADED_PARK_AFTER_CYCLES;
|
|
2888
2780
|
} else {
|
|
2889
|
-
// Warm, or a degraded stamped by the WATCHDOG (
|
|
2890
|
-
//
|
|
2891
|
-
//
|
|
2892
|
-
//
|
|
2893
|
-
// interrupted cycle re-armed short for the same reason.
|
|
2781
|
+
// Warm, or a degraded stamped by the WATCHDOG (stale-mid-flight) or by
|
|
2782
|
+
// an interrupted restore (restore-interrupted — a deploy rolled the
|
|
2783
|
+
// container; it says nothing about the repo): both mean a cycle must
|
|
2784
|
+
// RUN — this one.
|
|
2894
2785
|
// Counting those toward the streak would be self-fulfilling —
|
|
2895
2786
|
// each cycle that found the reason would park without attempting anything,
|
|
2896
2787
|
// and after three the resident would sit parked-degraded for 6h at a
|
|
@@ -2905,8 +2796,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2905
2796
|
...now,
|
|
2906
2797
|
idleSince: new Date(systemClock()).toISOString(),
|
|
2907
2798
|
} satisfies RepoFacts);
|
|
2908
|
-
|
|
2909
|
-
return { go: false, why: "idle" }; // the alarm's finally re-arms at IDLE_REFRESH_INTERVAL_S
|
|
2799
|
+
return { go: false, why: "idle" }; // the cron creates the next instance at IDLE_REFRESH_INTERVAL_S
|
|
2910
2800
|
}
|
|
2911
2801
|
if (facts.idleSince) {
|
|
2912
2802
|
const now = (await this.ctx.storage.get<RepoFacts>(FACTS_KEY)) ?? facts;
|
|
@@ -2924,9 +2814,9 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2924
2814
|
* the installation stays fresh, and a private one fails at the fetch
|
|
2925
2815
|
* below into a visible `degraded(github-unreachable: …)`. Returning early
|
|
2926
2816
|
* instead would freeze whatever state the resident was in — a public
|
|
2927
|
-
* repo the App is not installed on would sit
|
|
2928
|
-
*
|
|
2929
|
-
*
|
|
2817
|
+
* repo the App is not installed on would sit `degraded` forever with an
|
|
2818
|
+
* ever-staler mirror, because the App cannot mint for a repo it is not
|
|
2819
|
+
* installed on. A failed fetch
|
|
2930
2820
|
* is recorded here — `degraded` with its reason, the disk-full recovery
|
|
2931
2821
|
* when that is the cause — and answered `ok: false`. */
|
|
2932
2822
|
private async refreshFetch(
|
|
@@ -2957,7 +2847,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2957
2847
|
let sha: string;
|
|
2958
2848
|
try {
|
|
2959
2849
|
// Same mirror mutex as attach's fetch/worktree work: the
|
|
2960
|
-
// refresh
|
|
2850
|
+
// refresh cycle and an in-flight attach serialize instead of racing
|
|
2961
2851
|
// a prune against a worktree clone.
|
|
2962
2852
|
sha = (await this.fetchMirror({ ref: facts.defaultRef, cycle, token })).sha;
|
|
2963
2853
|
} catch (err) {
|
|
@@ -3042,9 +2932,9 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3042
2932
|
* them in the same write as the record — a wake never sees a half-updated
|
|
3043
2933
|
* pair; this is the cycle's own bookkeeping on a fresh read, and a
|
|
3044
2934
|
* superseded snapshot leaves the other writer's stamp alone), `warm`, then
|
|
3045
|
-
* the
|
|
3046
|
-
*
|
|
3047
|
-
* (item 55). */
|
|
2935
|
+
* the finished-ref reclamation the prune already informed (item 45) —
|
|
2936
|
+
* housekeeping, never a lifecycle flip. The disk sample is the instance's
|
|
2937
|
+
* own `measure` step (item 55), after its sweep. */
|
|
3048
2938
|
private async refreshComplete(
|
|
3049
2939
|
resource: string,
|
|
3050
2940
|
facts: RepoFacts,
|
|
@@ -3082,27 +2972,21 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3082
2972
|
} catch (err) {
|
|
3083
2973
|
console.log(`reclaim ${resource}: pass failed: ${errMsg(err)}`);
|
|
3084
2974
|
}
|
|
3085
|
-
// Item 55: the cycle's disk sample — what /residents, `repo list`, the
|
|
3086
|
-
// watchdog line and the next attach admission read. Housekeeping too.
|
|
3087
|
-
await this.measureDisk().catch((err) => console.log(`disk: measure failed: ${errMsg(err)}`));
|
|
3088
2975
|
}
|
|
3089
2976
|
|
|
3090
2977
|
/** What a cycle's throw means. A step killed from OUTSIDE (the container
|
|
3091
2978
|
* replaced under it — an image-changing deploy or a container stop; a
|
|
3092
2979
|
* Worker-only deploy leaves the container running and interrupts nothing)
|
|
3093
|
-
* is `refresh-interrupted`: not evidence about the repo
|
|
3094
|
-
*
|
|
3095
|
-
*
|
|
3096
|
-
*
|
|
3097
|
-
*
|
|
3098
|
-
*
|
|
3099
|
-
*
|
|
3100
|
-
* `refresh-failed: …`
|
|
3101
|
-
*
|
|
3102
|
-
*
|
|
3103
|
-
* `refresh-failed: …` shape. A full disk is a third class: `disk-full: …`,
|
|
3104
|
-
* never serviceable, and the one failure the resident can act on itself
|
|
3105
|
-
* (recoverFromDiskFull). */
|
|
2980
|
+
* is `refresh-interrupted`: not evidence about the repo, so it is never
|
|
2981
|
+
* recorded as `degraded` — the instance step throws it to the engine, whose
|
|
2982
|
+
* retry re-enters the same idempotent method (an unclassified kill would
|
|
2983
|
+
* instead cost the resident a `degraded(build-failed: exit 143 …)` until
|
|
2984
|
+
* the next cycle). Any other failure is the repo's own: `<step>-failed: …`
|
|
2985
|
+
* / `refresh-failed: …`. Non-StepErrors classify too — an SDK replacement
|
|
2986
|
+
* error can surface between steps — with the generic "refresh" step, whose
|
|
2987
|
+
* failure reason is the `refresh-failed: …` shape. A full disk is a third
|
|
2988
|
+
* class: `disk-full: …`, never serviceable, and the one failure the
|
|
2989
|
+
* resident can act on itself (recoverFromDiskFull). */
|
|
3106
2990
|
private async classifyCycleError(err: unknown): Promise<RefreshFailure> {
|
|
3107
2991
|
return err instanceof StepError
|
|
3108
2992
|
? await this.classifyFailure(err.step, err.message)
|
|
@@ -3127,56 +3011,38 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3127
3011
|
|
|
3128
3012
|
// -- the refresh cycle as a Workflow instance (item 7) --------------------------
|
|
3129
3013
|
//
|
|
3130
|
-
// `ResidentRefresh` (the Workflow entrypoint, refresh.ts) calls these
|
|
3131
|
-
// methods, one per step, through the DO stub. Each runs
|
|
3132
|
-
//
|
|
3014
|
+
// `ResidentRefresh` (the Workflow entrypoint, refresh.ts) calls these
|
|
3015
|
+
// methods, one per step, through the DO stub. Each runs one phase of the
|
|
3016
|
+
// cycle over the resident's rows, so a step the engine retries re-enters
|
|
3133
3017
|
// the same idempotent read-then-act method (item 22) and finds the work
|
|
3134
3018
|
// done. Inputs and answers are small facts — refs, shas, keys, a path, a
|
|
3135
3019
|
// word — never a payload and never a credential: the token is minted inside
|
|
3136
3020
|
// the step that needs it.
|
|
3137
3021
|
|
|
3138
|
-
/**
|
|
3022
|
+
/** The lifecycle row (LIFECYCLE_KEY): `workflow`, whatever a flagged rollout stored. */
|
|
3139
3023
|
async getLifecycle(): Promise<ResidentLifecycle> {
|
|
3140
3024
|
return lifecycleOf(await this.ctx.storage.get(LIFECYCLE_KEY));
|
|
3141
3025
|
}
|
|
3142
3026
|
|
|
3143
|
-
/**
|
|
3144
|
-
*
|
|
3145
|
-
*
|
|
3146
|
-
*
|
|
3147
|
-
|
|
3148
|
-
async setLifecycle(mode: ResidentLifecycle): Promise<{ lifecycle: ResidentLifecycle; refreshSchedules: number }> {
|
|
3149
|
-
const previous = await this.getLifecycle();
|
|
3027
|
+
/** Rewrite the lifecycle row (admin `/debug` `lifecycle`). `workflow` is
|
|
3028
|
+
* the one mode, so the op's only effect is to replace a stale `alarm`
|
|
3029
|
+
* value the flagged rollout left behind — the row then says what
|
|
3030
|
+
* `lifecycleOf` already reads. */
|
|
3031
|
+
async setLifecycle(mode: ResidentLifecycle): Promise<{ lifecycle: ResidentLifecycle }> {
|
|
3150
3032
|
await this.ctx.storage.put(LIFECYCLE_KEY, mode);
|
|
3151
|
-
|
|
3152
|
-
if (mode === "workflow") {
|
|
3153
|
-
this.deleteSchedules(REFRESH_CALLBACK);
|
|
3154
|
-
} else if (previous !== "alarm") {
|
|
3155
|
-
const { state } = await this.getStatus();
|
|
3156
|
-
const pending = await this.listSchedules(REFRESH_CALLBACK);
|
|
3157
|
-
if (state !== "onboarding" && state !== "down" && pending.length === 0) {
|
|
3158
|
-
await this.schedule(5, REFRESH_CALLBACK, resource);
|
|
3159
|
-
}
|
|
3160
|
-
}
|
|
3161
|
-
console.log(`lifecycle: ${resource} ${previous} → ${mode}`);
|
|
3162
|
-
return { lifecycle: mode, refreshSchedules: (await this.listSchedules(REFRESH_CALLBACK)).length };
|
|
3033
|
+
return { lifecycle: mode };
|
|
3163
3034
|
}
|
|
3164
3035
|
|
|
3165
3036
|
private async instanceRow(): Promise<RefreshInstanceRow> {
|
|
3166
3037
|
return (await this.ctx.storage.get<RefreshInstanceRow>(REFRESH_INSTANCE_KEY)) ?? { instance: null, skipped: null };
|
|
3167
3038
|
}
|
|
3168
3039
|
|
|
3169
|
-
/** The facts the cron's instance-creation decision
|
|
3170
|
-
*
|
|
3171
|
-
*
|
|
3040
|
+
/** The facts the cron's instance-creation decision and the admin
|
|
3041
|
+
* `refresh-now` op read (`shouldCreateRefreshInstance`,
|
|
3042
|
+
* `refreshCycleBlocked`): the state and when it last changed, idle mode,
|
|
3043
|
+
* the last instance's creation time and whether the engine still runs it. */
|
|
3172
3044
|
async refreshRow(): Promise<RefreshRow> {
|
|
3173
|
-
const map = await this.ctx.storage.get<unknown>([
|
|
3174
|
-
LIFECYCLE_KEY,
|
|
3175
|
-
STATE_KEY,
|
|
3176
|
-
UPDATED_KEY,
|
|
3177
|
-
FACTS_KEY,
|
|
3178
|
-
REFRESH_INSTANCE_KEY,
|
|
3179
|
-
]);
|
|
3045
|
+
const map = await this.ctx.storage.get<unknown>([STATE_KEY, UPDATED_KEY, FACTS_KEY, REFRESH_INSTANCE_KEY]);
|
|
3180
3046
|
const facts = map.get(FACTS_KEY) as RepoFacts | undefined;
|
|
3181
3047
|
const row = (map.get(REFRESH_INSTANCE_KEY) as RefreshInstanceRow | undefined) ?? { instance: null, skipped: null };
|
|
3182
3048
|
const epochMs = (iso: string | undefined): number | null => {
|
|
@@ -3184,7 +3050,6 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3184
3050
|
return Number.isFinite(t) ? t : null;
|
|
3185
3051
|
};
|
|
3186
3052
|
return {
|
|
3187
|
-
lifecycle: lifecycleOf(map.get(LIFECYCLE_KEY)),
|
|
3188
3053
|
state: (map.get(STATE_KEY) as ResidentState | undefined) ?? "down",
|
|
3189
3054
|
updatedAt: epochMs(map.get(UPDATED_KEY) as string | undefined),
|
|
3190
3055
|
idleSince: epochMs(facts?.idleSince),
|
|
@@ -3193,27 +3058,28 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3193
3058
|
};
|
|
3194
3059
|
}
|
|
3195
3060
|
|
|
3196
|
-
/**
|
|
3197
|
-
*
|
|
3198
|
-
*
|
|
3199
|
-
*
|
|
3200
|
-
*
|
|
3201
|
-
private async
|
|
3202
|
-
if (!id) return
|
|
3061
|
+
/** The engine's own word on the instance `id` — `queued`, `running`,
|
|
3062
|
+
* `complete`, `errored`, … — or null for an id the engine does not know (a
|
|
3063
|
+
* missing binding or a failed read answer the same). The one fact the
|
|
3064
|
+
* marker's age cannot give — a step between retry attempts holds no lease
|
|
3065
|
+
* and writes nothing — read from the engine, which knows. */
|
|
3066
|
+
private async instanceStatus(id: string | null): Promise<string | null> {
|
|
3067
|
+
if (!id) return null;
|
|
3203
3068
|
try {
|
|
3204
|
-
|
|
3205
|
-
return (
|
|
3206
|
-
status === "queued" ||
|
|
3207
|
-
status === "running" ||
|
|
3208
|
-
status === "paused" ||
|
|
3209
|
-
status === "waiting" ||
|
|
3210
|
-
status === "waitingForPause"
|
|
3211
|
-
);
|
|
3069
|
+
return (await (await this.env.RESIDENT_REFRESH.get(id)).status()).status;
|
|
3212
3070
|
} catch {
|
|
3213
|
-
return
|
|
3071
|
+
return null;
|
|
3214
3072
|
}
|
|
3215
3073
|
}
|
|
3216
3074
|
|
|
3075
|
+
/** Whether the engine still runs `id`: queued, running, paused or waiting.
|
|
3076
|
+
* Anything else — a finished or failed instance, an unknown id — is not a
|
|
3077
|
+
* live cycle, and the marker's age then decides, as before. */
|
|
3078
|
+
private async instanceRunning(id: string | null): Promise<boolean> {
|
|
3079
|
+
const status = await this.instanceStatus(id);
|
|
3080
|
+
return status !== null && INSTANCE_LIVE_STATUSES.has(status);
|
|
3081
|
+
}
|
|
3082
|
+
|
|
3217
3083
|
/** The cron created an instance for this resident. */
|
|
3218
3084
|
async recordRefreshInstance(id: string, createdAtMs: number): Promise<void> {
|
|
3219
3085
|
const row = await this.instanceRow();
|
|
@@ -3288,28 +3154,39 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3288
3154
|
} satisfies RefreshInstanceRow);
|
|
3289
3155
|
}
|
|
3290
3156
|
|
|
3291
|
-
/** Run one step of the refresh instance
|
|
3292
|
-
*
|
|
3293
|
-
*
|
|
3157
|
+
/** Run one step of the refresh instance: counted in flight once past the
|
|
3158
|
+
* entry gates (so an attach-path reconcileImage never stops the container
|
|
3159
|
+
* under it — while the gates themselves, `isIdle`, `reconcileImage("refresh")`
|
|
3160
|
+
* and the disk-full recycle, must not see the probing cycle as an operation
|
|
3161
|
+
* in flight, or no resident would ever park, restart a stale image or
|
|
3162
|
+
* recycle a full disk; the step is handed `cycle.count` and calls it once
|
|
3163
|
+
* its gates have passed — the fetch step after `refreshGate`, every other
|
|
3164
|
+
* step first thing), under a step trace the instance grafts on its root,
|
|
3294
3165
|
* its outcome on the instance row for `/status`. A step killed from outside
|
|
3295
|
-
* (the container replaced under it)
|
|
3166
|
+
* (the container replaced under it), or a gate that stopped the container
|
|
3167
|
+
* on purpose (`CycleRestartError`), is thrown to the engine, whose retry
|
|
3296
3168
|
* re-enters the same idempotent method — the row stays `refreshing`, never
|
|
3297
3169
|
* `degraded`, and a `refreshing` younger than the stale bound keeps the
|
|
3298
3170
|
* cron from creating a second instance meanwhile. A failure of the repo's
|
|
3299
|
-
* own is recorded
|
|
3300
|
-
*
|
|
3301
|
-
*
|
|
3171
|
+
* own is recorded — `degraded` with the reason, the last snapshot still
|
|
3172
|
+
* serving — and answered `failed`, which ends the cycle; the next cron
|
|
3173
|
+
* firing starts the next one from that state. */
|
|
3302
3174
|
private async runInstanceStep<T>(
|
|
3303
3175
|
instance: string,
|
|
3304
3176
|
step: string,
|
|
3305
|
-
fn: () => Promise<InstanceStepResult<T>>,
|
|
3177
|
+
fn: (cycle: { count: () => void }) => Promise<InstanceStepResult<T>>,
|
|
3306
3178
|
): Promise<InstanceStepAnswer<T>> {
|
|
3307
3179
|
const startedAt = systemClock();
|
|
3308
3180
|
const trace = createStepTrace(startedAt);
|
|
3309
|
-
|
|
3181
|
+
let counted = false;
|
|
3182
|
+
const count = () => {
|
|
3183
|
+
if (counted) return;
|
|
3184
|
+
counted = true;
|
|
3185
|
+
this.refreshesInFlight++;
|
|
3186
|
+
};
|
|
3310
3187
|
let outcome = "done";
|
|
3311
3188
|
try {
|
|
3312
|
-
const result = await this.stepTrace.run(trace, fn);
|
|
3189
|
+
const result = await this.stepTrace.run(trace, () => fn({ count }));
|
|
3313
3190
|
if (result.status !== "done") {
|
|
3314
3191
|
outcome = result.status === "stopped" ? `stopped (${result.why})` : `failed (${result.reason})`;
|
|
3315
3192
|
await this.clearInstanceLease(instance);
|
|
@@ -3322,20 +3199,27 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3322
3199
|
await this.clearInstanceLease(instance);
|
|
3323
3200
|
return { status: "failed", reason: err.message, startedAt, trace: trace.steps() };
|
|
3324
3201
|
}
|
|
3202
|
+
if (err instanceof CycleRestartError) {
|
|
3203
|
+
outcome = `restarting (${err.why}) — the engine retries`;
|
|
3204
|
+
console.log(`refresh instance ${instance}: ${step} ${err.message}`);
|
|
3205
|
+
throw err;
|
|
3206
|
+
}
|
|
3325
3207
|
const failure = await this.classifyCycleError(err);
|
|
3326
3208
|
if (failure.interrupted) {
|
|
3327
3209
|
outcome = `interrupted (${failure.reason}) — the engine retries`;
|
|
3328
3210
|
console.log(`refresh instance ${instance}: ${step} interrupted — ${failure.reason.slice(0, 400)}; retrying`);
|
|
3329
3211
|
throw err;
|
|
3330
3212
|
}
|
|
3331
|
-
|
|
3213
|
+
// The disk-full recovery excludes this cycle from its in-flight count only
|
|
3214
|
+
// when the cycle counted itself: a failure inside the gates (before
|
|
3215
|
+
// `cycle.count()`) contributed nothing, and excluding it anyway would read
|
|
3216
|
+
// one real run as none and recycle the container under it.
|
|
3217
|
+
await this.refreshFailed(failure, counted ? 1 : 0);
|
|
3332
3218
|
await this.clearInstanceLease(instance);
|
|
3333
3219
|
outcome = `failed (${failure.reason})`;
|
|
3334
3220
|
return { status: "failed", reason: failure.reason, startedAt, trace: trace.steps() };
|
|
3335
3221
|
} finally {
|
|
3336
|
-
this.refreshesInFlight--;
|
|
3337
|
-
// The gates set this for the alarm's finally; no alarm runs on this path.
|
|
3338
|
-
this.rearmOutcome = "normal";
|
|
3222
|
+
if (counted) this.refreshesInFlight--;
|
|
3339
3223
|
await this.recordInstanceStep(instance, step, outcome).catch((err) =>
|
|
3340
3224
|
console.log(`refresh instance ${instance}: recording ${step} failed: ${errMsg(err)}`),
|
|
3341
3225
|
);
|
|
@@ -3350,12 +3234,17 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3350
3234
|
resource: string;
|
|
3351
3235
|
instance: string;
|
|
3352
3236
|
}): Promise<InstanceStepAnswer<RefreshFetchFacts>> {
|
|
3353
|
-
return this.runInstanceStep<RefreshFetchFacts>(input.instance, "fetch", async () => {
|
|
3237
|
+
return this.runInstanceStep<RefreshFetchFacts>(input.instance, "fetch", async (cycle) => {
|
|
3354
3238
|
const before = await this.getStatus();
|
|
3355
3239
|
// down stays down (a rebuild is the escape hatch); onboarding is owned by provisioning.
|
|
3356
3240
|
if (before.state === "onboarding" || before.state === "down") return { status: "stopped", why: "not-serving" };
|
|
3357
3241
|
const gate = await this.refreshGate(input.resource);
|
|
3358
3242
|
if (!gate.go) return { status: "stopped", why: gate.why };
|
|
3243
|
+
// Past the gates the cycle mutates the mirror and checkout: count it in
|
|
3244
|
+
// flight from here — never before, or the gates above (the idle park,
|
|
3245
|
+
// the image reconcile, the disk-full recycle) would have seen this
|
|
3246
|
+
// cycle as a live operation and never fired.
|
|
3247
|
+
cycle.count();
|
|
3359
3248
|
const { record, facts } = gate;
|
|
3360
3249
|
const holder = this.nextHolder();
|
|
3361
3250
|
await this.recordInFlight("refresh", holder, REFRESH_CYCLE_LEASE_MS, "refresh");
|
|
@@ -3394,7 +3283,8 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3394
3283
|
sha: string;
|
|
3395
3284
|
lockfileKey: string;
|
|
3396
3285
|
}): Promise<InstanceStepAnswer<{ entry: string | null }>> {
|
|
3397
|
-
return this.runInstanceStep<{ entry: string | null }>(input.instance, "install", async () => {
|
|
3286
|
+
return this.runInstanceStep<{ entry: string | null }>(input.instance, "install", async (cycle) => {
|
|
3287
|
+
cycle.count();
|
|
3398
3288
|
const record = await this.registry().getRecord(input.resource);
|
|
3399
3289
|
if (!record) return { status: "stopped", why: "offboarded" };
|
|
3400
3290
|
const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
|
|
@@ -3414,7 +3304,8 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3414
3304
|
lockfileKey: string;
|
|
3415
3305
|
depsEntry: string | null;
|
|
3416
3306
|
}): Promise<InstanceStepAnswer<{ ran: boolean; why: string }>> {
|
|
3417
|
-
return this.runInstanceStep<{ ran: boolean; why: string }>(input.instance, "build", async () => {
|
|
3307
|
+
return this.runInstanceStep<{ ran: boolean; why: string }>(input.instance, "build", async (cycle) => {
|
|
3308
|
+
cycle.count();
|
|
3418
3309
|
const record = await this.registry().getRecord(input.resource);
|
|
3419
3310
|
if (!record) return { status: "stopped", why: "offboarded" };
|
|
3420
3311
|
const built = await this.runBuild({
|
|
@@ -3432,9 +3323,9 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3432
3323
|
|
|
3433
3324
|
/** Step `snapshot`: the stamped pair to R2 (`snapshot` finds a record at the
|
|
3434
3325
|
* stamp done and answers `superseded` to another writer, never a throw),
|
|
3435
|
-
* then the cycle's completion — facts, `warm`, the reclamation
|
|
3436
|
-
*
|
|
3437
|
-
*
|
|
3326
|
+
* then the cycle's completion — facts, `warm`, the reclamation — and the
|
|
3327
|
+
* cycle lease released. An `unchanged` cycle skips the archive and still
|
|
3328
|
+
* completes. */
|
|
3438
3329
|
async refreshInstanceSnapshot(input: {
|
|
3439
3330
|
resource: string;
|
|
3440
3331
|
instance: string;
|
|
@@ -3444,7 +3335,8 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3444
3335
|
action: RefreshPlan["action"];
|
|
3445
3336
|
mintError: string | null;
|
|
3446
3337
|
}): Promise<InstanceStepAnswer<{ committed: boolean }>> {
|
|
3447
|
-
return this.runInstanceStep<{ committed: boolean }>(input.instance, "snapshot", async () => {
|
|
3338
|
+
return this.runInstanceStep<{ committed: boolean }>(input.instance, "snapshot", async (cycle) => {
|
|
3339
|
+
cycle.count();
|
|
3448
3340
|
const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
|
|
3449
3341
|
if (!facts) return { status: "stopped", why: "no-facts" };
|
|
3450
3342
|
const committed =
|
|
@@ -3480,11 +3372,50 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3480
3372
|
});
|
|
3481
3373
|
}
|
|
3482
3374
|
|
|
3483
|
-
/**
|
|
3484
|
-
*
|
|
3485
|
-
*
|
|
3486
|
-
|
|
3487
|
-
|
|
3375
|
+
/** A housekeeping step never flips lifecycle state: its failure is a log
|
|
3376
|
+
* line and a `done` answer naming it (`result` null) — except a step killed
|
|
3377
|
+
* from outside, which the engine retries like any other. */
|
|
3378
|
+
private async housekeeping<T>(
|
|
3379
|
+
step: string,
|
|
3380
|
+
work: () => Promise<T>,
|
|
3381
|
+
): Promise<InstanceStepResult<{ result: T | null; error: string | null }>> {
|
|
3382
|
+
try {
|
|
3383
|
+
return { status: "done", result: await work(), error: null };
|
|
3384
|
+
} catch (err) {
|
|
3385
|
+
const failure = await this.classifyCycleError(err);
|
|
3386
|
+
if (failure.interrupted) throw err;
|
|
3387
|
+
console.log(`refresh instance: ${step} failed — ${failure.reason.slice(0, 400)}`);
|
|
3388
|
+
return { status: "done", result: null, error: residentText(failure.reason) };
|
|
3389
|
+
}
|
|
3390
|
+
}
|
|
3391
|
+
|
|
3392
|
+
/** Step `sweep`: the worktree inactivity sweep (item 23) — TTL eviction and
|
|
3393
|
+
* the clean-idle release. No gate runs here, so the step counts its cycle
|
|
3394
|
+
* first thing, like install, build and snapshot. Idempotent by shape: a
|
|
3395
|
+
* second call finds the bindings it evicted already evicted and nothing
|
|
3396
|
+
* else past its cutoffs. */
|
|
3397
|
+
async refreshInstanceSweep(input: {
|
|
3398
|
+
resource: string;
|
|
3399
|
+
instance: string;
|
|
3400
|
+
}): Promise<InstanceStepAnswer<{ result: { evicted: string[]; kept: number } | null; error: string | null }>> {
|
|
3401
|
+
return this.runInstanceStep(input.instance, "sweep", async (cycle) => {
|
|
3402
|
+
cycle.count();
|
|
3403
|
+
return this.housekeeping("sweep", () => this.sweepWorktrees(input.resource));
|
|
3404
|
+
});
|
|
3405
|
+
}
|
|
3406
|
+
|
|
3407
|
+
/** Step `measure`: the disk sample (item 55) — one `df` + one `du`, written
|
|
3408
|
+
* over the last; a second call takes the same sample again. Counts its
|
|
3409
|
+
* cycle first thing, like every step past the gates. Never wakes a slept
|
|
3410
|
+
* container (`measureDisk` answers null, `measured: false`). */
|
|
3411
|
+
async refreshInstanceMeasure(input: {
|
|
3412
|
+
instance: string;
|
|
3413
|
+
}): Promise<InstanceStepAnswer<{ result: { measured: boolean } | null; error: string | null }>> {
|
|
3414
|
+
return this.runInstanceStep(input.instance, "measure", async (cycle) => {
|
|
3415
|
+
cycle.count();
|
|
3416
|
+
return this.housekeeping("measure", async () => ({ measured: (await this.measureDisk()) !== null }));
|
|
3417
|
+
});
|
|
3418
|
+
}
|
|
3488
3419
|
|
|
3489
3420
|
/** Idle = no live binding attached within IDLE_AFTER_S AND (when the
|
|
3490
3421
|
* runtime is up) no live tree is dirty. Bindings are storage; dirtiness
|
|
@@ -3538,14 +3469,15 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3538
3469
|
return classifyRefreshFailure({ step, message, freeKiB: await this.freeKiB() });
|
|
3539
3470
|
}
|
|
3540
3471
|
|
|
3541
|
-
/** The disk is a cache: stop the container so the next
|
|
3542
|
-
* mirror + checkout from R2 onto an empty disk — the same wake
|
|
3543
|
-
* platform sleep. Only when the pure plan allows it: nothing in
|
|
3544
|
-
* (`selfInFlight` excludes the calling refresh cycle from the
|
|
3545
|
-
* live tree clean, and no recycle within the cooldown. A
|
|
3546
|
-
* written to `lastRefreshError` with its why, so
|
|
3547
|
-
* operator must do; the `degraded` reason stays
|
|
3548
|
-
|
|
3472
|
+
/** The disk is a cache: stop the container so the next cycle attempt
|
|
3473
|
+
* restores mirror + checkout from R2 onto an empty disk — the same wake
|
|
3474
|
+
* path as a platform sleep. Only when the pure plan allows it: nothing in
|
|
3475
|
+
* flight (`selfInFlight` excludes the calling refresh cycle from the
|
|
3476
|
+
* count), every live tree clean, and no recycle within the cooldown. A
|
|
3477
|
+
* refused recycle is written to `lastRefreshError` with its why, so
|
|
3478
|
+
* `/residents` says what an operator must do; the `degraded` reason stays
|
|
3479
|
+
* the clean `disk-full: …`. Answers whether the container was recycled. */
|
|
3480
|
+
private async recoverFromDiskFull(reason: string, selfInFlight: number): Promise<boolean> {
|
|
3549
3481
|
const lastRecycleAt = await this.ctx.storage.get<number>(DISK_FULL_RECYCLE_KEY);
|
|
3550
3482
|
const plan = planDiskFullRecovery({
|
|
3551
3483
|
now: systemClock(),
|
|
@@ -3556,16 +3488,16 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3556
3488
|
if (plan.action === "wait") {
|
|
3557
3489
|
console.log(`disk-full: container kept — ${plan.why}`);
|
|
3558
3490
|
await this.recordRefreshError(`${reason} — container kept: ${plan.why}`);
|
|
3559
|
-
return;
|
|
3491
|
+
return false;
|
|
3560
3492
|
}
|
|
3561
3493
|
console.log(
|
|
3562
|
-
`disk-full: recycling the container — the next
|
|
3494
|
+
`disk-full: recycling the container — the next cycle restores mirror + checkout from R2 onto an empty disk (${reason})`,
|
|
3563
3495
|
);
|
|
3564
3496
|
await this.ctx.storage.put(DISK_FULL_RECYCLE_KEY, systemClock());
|
|
3565
|
-
await this.recordRefreshError(`${reason} — container recycled; restoring from R2 on the next
|
|
3497
|
+
await this.recordRefreshError(`${reason} — container recycled; restoring from R2 on the next cycle`);
|
|
3566
3498
|
this.swapIncarnation(); // deliberate incarnation swap
|
|
3567
3499
|
await this.stop().catch((err) => console.log(`disk-full: stop failed: ${errMsg(err)}`));
|
|
3568
|
-
|
|
3500
|
+
return true;
|
|
3569
3501
|
}
|
|
3570
3502
|
|
|
3571
3503
|
// -- disk budget (docs/reference/specs/resident-repos.md item 55) -------------------------
|
|
@@ -3625,18 +3557,6 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3625
3557
|
return sample;
|
|
3626
3558
|
}
|
|
3627
3559
|
|
|
3628
|
-
/** Schedule callback: the deferred measurement an attach/detach/sweep arms. */
|
|
3629
|
-
async onDiskMeasure(_payload: string): Promise<void> {
|
|
3630
|
-
await this.measureDisk().catch((err) => console.log(`disk: measure failed: ${errMsg(err)}`));
|
|
3631
|
-
}
|
|
3632
|
-
|
|
3633
|
-
/** Arm one deferred measurement (at most one pending): the attach's hot path
|
|
3634
|
-
* pays a `df`, not the `du`. */
|
|
3635
|
-
private async scheduleDiskMeasure(resource: string): Promise<void> {
|
|
3636
|
-
if ((await this.listSchedules(DISK_MEASURE_CALLBACK)).length > 0) return;
|
|
3637
|
-
await this.schedule(DISK_MEASURE_DELAY_S, DISK_MEASURE_CALLBACK, resource);
|
|
3638
|
-
}
|
|
3639
|
-
|
|
3640
3560
|
/** `{usedKiB, totalKiB, at}` of the last sample for the watchdog line (storage
|
|
3641
3561
|
* only — the watchdog never touches the container). */
|
|
3642
3562
|
private async diskGauge(): Promise<{ usedKiB: number; totalKiB: number; freeKiB: number; at: string } | null> {
|
|
@@ -3781,9 +3701,10 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3781
3701
|
|
|
3782
3702
|
/** Refresh-on-attach, BOUNDED: if the resident was idle (or the last
|
|
3783
3703
|
* refresh is older than the active cadence), fetch the mirror now — seconds,
|
|
3784
|
-
* under the mirror lock — so the ref this thread binds is current,
|
|
3785
|
-
* idle mode,
|
|
3786
|
-
* moved: minutes)
|
|
3704
|
+
* under the mirror lock — so the ref this thread binds is current, and
|
|
3705
|
+
* clear idle mode, which puts the full refresh cycle (checkout rebuild if
|
|
3706
|
+
* main moved: minutes) on the cron's awake cadence — the next firing
|
|
3707
|
+
* creates its instance, within one bucket. The attach never waits on an
|
|
3787
3708
|
* install/build, so a wake cannot become a cold-fallback generator. */
|
|
3788
3709
|
private async refreshIfStale(resource: string): Promise<void> {
|
|
3789
3710
|
const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
|
|
@@ -3798,7 +3719,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3798
3719
|
// Bounded like attach's own clone section: a full checkout rebuild
|
|
3799
3720
|
// holding the mutex must not stall a wake attach for minutes — on
|
|
3800
3721
|
// expiry (MirrorBusyError) proceed on the last mirror, same as a failed
|
|
3801
|
-
// fetch. The
|
|
3722
|
+
// fetch. The next refresh instance repays the staleness.
|
|
3802
3723
|
await this.withMirrorLock(
|
|
3803
3724
|
() =>
|
|
3804
3725
|
this.gitWithCred(
|
|
@@ -3822,7 +3743,6 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3822
3743
|
const { idleSince: _woke, ...awake } = fresh;
|
|
3823
3744
|
await this.ctx.storage.put(FACTS_KEY, awake satisfies RepoFacts);
|
|
3824
3745
|
}
|
|
3825
|
-
await this.armRefresh(resource, 1); // full cycle now, in the background; it re-arms at the active cadence
|
|
3826
3746
|
}
|
|
3827
3747
|
|
|
3828
3748
|
/** Pool users live in the IMAGE (Dockerfile useradd loop) while THREAD_USERS
|
|
@@ -3851,20 +3771,26 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3851
3771
|
return true;
|
|
3852
3772
|
}
|
|
3853
3773
|
|
|
3854
|
-
// -- watchdog (the sparse cron
|
|
3855
|
-
|
|
3856
|
-
/** One watchdog pass over this resident (invoked by the Worker cron):
|
|
3857
|
-
*
|
|
3858
|
-
*
|
|
3859
|
-
*
|
|
3860
|
-
*
|
|
3861
|
-
*
|
|
3862
|
-
*
|
|
3774
|
+
// -- watchdog (the sparse cron; it re-arms nothing) --------------------------
|
|
3775
|
+
|
|
3776
|
+
/** One watchdog pass over this resident (invoked by the Worker cron): time
|
|
3777
|
+
* out an onboarding stuck past its budget → down(provision-timeout) + cap
|
|
3778
|
+
* slot release; auto-rebuild a resident stuck down on unusable snapshots
|
|
3779
|
+
* (one strike per pass, rebuild at AUTO_REBUILD_AFTER_STRIKES); name a
|
|
3780
|
+
* `refreshing`/`restoring` marker older than STALE_MIDFLIGHT_MS that no
|
|
3781
|
+
* live lease and no running instance stands behind —
|
|
3782
|
+
* `degraded(stale-mid-flight: …)` carrying the last instance's id and the
|
|
3783
|
+
* engine's word on it, so a failed instance is visible by name on
|
|
3784
|
+
* `/status` (item 9) — and the next instance the cron creates (this very
|
|
3785
|
+
* pass) normalizes it. Storage and engine-status reads only: containers
|
|
3786
|
+
* are started by the instances' steps, never in this pass. The refresh row
|
|
3787
|
+
* the cron's instance-creation decision reads rides on the answer, read
|
|
3788
|
+
* after the check settled the state. */
|
|
3863
3789
|
async watchdogCheck(): Promise<{
|
|
3864
3790
|
resource: string;
|
|
3865
3791
|
state: ResidentState;
|
|
3866
3792
|
reason: string;
|
|
3867
|
-
action: "none" | "
|
|
3793
|
+
action: "none" | "provision-timed-out" | "auto-rebuilt";
|
|
3868
3794
|
/** Item 55: the last disk sample's gauge, for the watchdog's status line. */
|
|
3869
3795
|
disk: { usedKiB: number; totalKiB: number; freeKiB: number; at: string } | null;
|
|
3870
3796
|
/** Item 7: what the cron's instance-creation decision reads, after the check above settled the state. */
|
|
@@ -3878,7 +3804,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3878
3804
|
resource: string;
|
|
3879
3805
|
state: ResidentState;
|
|
3880
3806
|
reason: string;
|
|
3881
|
-
action: "none" | "
|
|
3807
|
+
action: "none" | "provision-timed-out" | "auto-rebuilt";
|
|
3882
3808
|
}> {
|
|
3883
3809
|
const resource = (await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "";
|
|
3884
3810
|
const status = await this.getStatus();
|
|
@@ -3893,8 +3819,8 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3893
3819
|
}
|
|
3894
3820
|
if (status.state === "down") {
|
|
3895
3821
|
// Auto-rebuild escape hatch: only rehydration-flavored downs — the
|
|
3896
|
-
// snapshots themselves are the problem, and down
|
|
3897
|
-
// without this the resident would stay down forever.
|
|
3822
|
+
// snapshots themselves are the problem, and a down resident runs no
|
|
3823
|
+
// cycle, so without this the resident would stay down forever.
|
|
3898
3824
|
if (REHYDRATION_FAILURE_RE.test(status.reason)) {
|
|
3899
3825
|
const strikes = ((await this.ctx.storage.get<number>(REBUILD_STRIKES_KEY)) ?? 0) + 1;
|
|
3900
3826
|
if (strikes >= AUTO_REBUILD_AFTER_STRIKES) {
|
|
@@ -3914,51 +3840,16 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3914
3840
|
// Any serving state clears accumulated strikes (a recovery must reset the
|
|
3915
3841
|
// counter, or an unrelated later down inherits stale strikes).
|
|
3916
3842
|
await this.ctx.storage.delete(REBUILD_STRIKES_KEY);
|
|
3917
|
-
// Item 7: a resident on the Workflow lifecycle has no chain to re-arm —
|
|
3918
|
-
// its cycles are the instances the cron creates. The sweep re-arm below
|
|
3919
|
-
// is housekeeping either way; the two refresh re-arms are the alarm's.
|
|
3920
|
-
const chained = (await this.getLifecycle()) === "alarm";
|
|
3921
|
-
|
|
3922
|
-
// The sweep chain has the same failure mode as the refresh chain (a DO
|
|
3923
|
-
// eviction mid-callback kills the self-rescheduling), but nothing re-armed
|
|
3924
|
-
// it: only an attach did, so a resident with live bindings and no traffic
|
|
3925
|
-
// never swept again (idle bindings sat for hours with no sweep).
|
|
3926
|
-
// Re-arm at +5s whenever live bindings exist and none is pending. Not a
|
|
3927
|
-
// lifecycle event — the sweep is housekeeping, no state flip. Runs BEFORE
|
|
3928
|
-
// the stale-mid-flight check so that branch's early return never skips it.
|
|
3929
|
-
const bindings = await this.ctx.storage.list<ThreadBinding>({ prefix: THREAD_KEY_PREFIX });
|
|
3930
|
-
const liveBindings = [...bindings.values()].some((b) => !b.evicted && b.user);
|
|
3931
|
-
// `sweepInFlight` is the explicit guard against arming a second chain while
|
|
3932
|
-
// a sweep is executing (its schedule row also stays listed until the callback
|
|
3933
|
-
// resolves, but that is a library detail we do not lean on).
|
|
3934
|
-
if (liveBindings && !this.sweepInFlight) {
|
|
3935
|
-
const pendingSweeps = await this.listSchedules(SWEEP_CALLBACK);
|
|
3936
|
-
// Config drift: a row armed by OLDER code (e.g. a daily sweep from
|
|
3937
|
-
// before the cadence shortened, due 24h out) is still honored by the runtime, so a
|
|
3938
|
-
// shorter SWEEP_INTERVAL_S never takes effect until it fires. Treat a row
|
|
3939
|
-
// due further out than the current interval (+ slack) as stale and
|
|
3940
|
-
// replace it, so a deploy that shortens the cadence applies within one
|
|
3941
|
-
// watchdog pass rather than after the old delay elapses.
|
|
3942
|
-
const nowS = Math.floor(systemClock() / 1000);
|
|
3943
|
-
const drifted = pendingSweeps.some((row) => (row.time ?? 0) - nowS > SWEEP_INTERVAL_S + SWEEP_DRIFT_SLACK_S);
|
|
3944
|
-
// Re-check the guard: listSchedules yielded, and a sweep that started
|
|
3945
|
-
// meanwhile owns the row its own `finally` is about to arm.
|
|
3946
|
-
if ((pendingSweeps.length === 0 || drifted) && !this.sweepInFlight) {
|
|
3947
|
-
this.deleteSchedules(SWEEP_CALLBACK);
|
|
3948
|
-
await this.schedule(5, SWEEP_CALLBACK, resource);
|
|
3949
|
-
// Disjoint by construction: inside this branch, a non-empty list implies `drifted`.
|
|
3950
|
-
console.log(
|
|
3951
|
-
`watchdog ${resource}: sweep ${pendingSweeps.length === 0 ? "chain was dead" : "row was due beyond the current interval (config drift)"} with live bindings — re-armed`,
|
|
3952
|
-
);
|
|
3953
|
-
}
|
|
3954
|
-
}
|
|
3955
3843
|
|
|
3956
3844
|
// A mid-flight state older than STALE_MIDFLIGHT_MS with no cycle or restore
|
|
3957
|
-
// actually running is a marker orphaned by
|
|
3958
|
-
// by a deploy, platform restart
|
|
3959
|
-
//
|
|
3960
|
-
// the bot's warm-gate keeps sending
|
|
3961
|
-
//
|
|
3845
|
+
// actually running is a marker orphaned by a cycle that died — a DO evicted
|
|
3846
|
+
// by a deploy, a platform restart, an instance out of retries mid-step.
|
|
3847
|
+
// Left alone it is permanent — the idle gate only parks from `warm`, and
|
|
3848
|
+
// nothing else would ever rewrite it, so the bot's warm-gate keeps sending
|
|
3849
|
+
// runs cold. Mark it degraded (visible — named degradation, never a
|
|
3850
|
+
// stall), naming the instance the row last recorded and the engine's word
|
|
3851
|
+
// on it; the marker is no longer `refreshing`, so the instance the cron
|
|
3852
|
+
// creates in this same pass normalizes it.
|
|
3962
3853
|
if (status.state === "refreshing" || status.state === "restoring") {
|
|
3963
3854
|
const updatedAt = Date.parse((await this.ctx.storage.get<string>(UPDATED_KEY)) ?? "") || 0;
|
|
3964
3855
|
// Who holds what comes from the in-flight row (item 22), not from this
|
|
@@ -3980,49 +3871,32 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3980
3871
|
if (again.state !== status.state || liveAgain.refresh || liveAgain.hydration) {
|
|
3981
3872
|
return { resource, ...again, action: "none" };
|
|
3982
3873
|
}
|
|
3983
|
-
|
|
3984
|
-
|
|
3985
|
-
|
|
3986
|
-
// nothing on the other end is still writing (the container it talked
|
|
3987
|
-
// to is gone). Each clear is compared against the holder just read:
|
|
3988
|
-
// a cycle that recorded a fresh lease between that read and this
|
|
3989
|
-
// delete keeps it, the way a release never deletes another holder's row.
|
|
3990
|
-
this.hydration = null;
|
|
3991
|
-
if (!chained && (await this.instanceRunning((await this.instanceRow()).instance?.id ?? null))) {
|
|
3874
|
+
const last = (await this.instanceRow()).instance?.id ?? null;
|
|
3875
|
+
const engine = await this.instanceStatus(last);
|
|
3876
|
+
if (engine !== null && INSTANCE_LIVE_STATUSES.has(engine)) {
|
|
3992
3877
|
// The engine still runs the recorded instance — a step between retry
|
|
3993
3878
|
// attempts, holding no lease and writing no state. Not stale: leave
|
|
3994
3879
|
// the marker, create nothing (the cron's decision reads the same fact).
|
|
3995
3880
|
return { resource, ...status, action: "none" };
|
|
3996
3881
|
}
|
|
3882
|
+
// Drop the dead hydration reference and the dead leases so the next
|
|
3883
|
+
// cycle's ensureHydrated starts a fresh restore instead of awaiting a
|
|
3884
|
+
// promise that will never settle. Safe: past the bound nothing on the
|
|
3885
|
+
// other end is still writing (the container it talked to is gone).
|
|
3886
|
+
// Each clear is compared against the holder just read: a cycle that
|
|
3887
|
+
// recorded a fresh lease between that read and this delete keeps it,
|
|
3888
|
+
// the way a release never deletes another holder's row.
|
|
3889
|
+
this.hydration = null;
|
|
3997
3890
|
if (rowAgain.refresh) await this.clearInFlight("refresh", rowAgain.refresh.holder);
|
|
3998
3891
|
if (rowAgain.hydration) await this.clearInFlight("hydration", rowAgain.hydration.holder);
|
|
3999
|
-
|
|
4000
|
-
|
|
4001
|
-
|
|
4002
|
-
|
|
4003
|
-
const reason = `stale-mid-flight: ${status.state} since ${new Date(updatedAt).toISOString()} with no cycle running; the next refresh instance normalizes it`;
|
|
4004
|
-
await this.setResidentState("degraded", reason);
|
|
4005
|
-
return { resource, state: "degraded", reason, action: "none" };
|
|
4006
|
-
}
|
|
4007
|
-
const reason = `stale-mid-flight: ${status.state} since ${new Date(updatedAt).toISOString()} with no cycle running; re-armed by watchdog`;
|
|
3892
|
+
const instance = last
|
|
3893
|
+
? `the last instance ${last} is ${engine ?? "unknown to the engine"}`
|
|
3894
|
+
: "no instance recorded";
|
|
3895
|
+
const reason = `stale-mid-flight: ${status.state} since ${new Date(updatedAt).toISOString()} with no cycle running; ${instance}; the next refresh instance normalizes it`;
|
|
4008
3896
|
await this.setResidentState("degraded", reason);
|
|
4009
|
-
|
|
4010
|
-
await this.schedule(5, REFRESH_CALLBACK, resource);
|
|
4011
|
-
return { resource, state: "degraded", reason, action: "rearmed" };
|
|
3897
|
+
return { resource, state: "degraded", reason, action: "none" };
|
|
4012
3898
|
}
|
|
4013
3899
|
}
|
|
4014
|
-
|
|
4015
|
-
// No chain to be dead under the Workflow lifecycle (item 7).
|
|
4016
|
-
if (!chained) return { resource, ...status, action: "none" };
|
|
4017
|
-
const pending = await this.listSchedules(REFRESH_CALLBACK);
|
|
4018
|
-
if (pending.length === 0) {
|
|
4019
|
-
await this.schedule(5, REFRESH_CALLBACK, resource);
|
|
4020
|
-
const reason = "alarm-missed: refresh chain was dead; re-armed by watchdog";
|
|
4021
|
-
// An in-flight restore owns its own state; everything else is visibly
|
|
4022
|
-
// degraded until the re-armed refresh succeeds.
|
|
4023
|
-
if (status.state !== "restoring") await this.setResidentState("degraded", reason);
|
|
4024
|
-
return { resource, state: "degraded", reason, action: "rearmed" };
|
|
4025
|
-
}
|
|
4026
3900
|
return { resource, ...status, action: "none" };
|
|
4027
3901
|
}
|
|
4028
3902
|
|
|
@@ -4197,7 +4071,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
4197
4071
|
}
|
|
4198
4072
|
// From here the attach may hold the mirror lock through clone/install:
|
|
4199
4073
|
// count it so a concurrent refresh-cycle reconcileImage never stops the
|
|
4200
|
-
// container under it (and isIdle never parks the
|
|
4074
|
+
// container under it (and isIdle never parks the cycle mid-attach).
|
|
4201
4075
|
this.attachesInFlight++;
|
|
4202
4076
|
try {
|
|
4203
4077
|
return await this.attachThreadBody(threadKey, refHint, readonly, wantSha, resourceId, t0, record);
|
|
@@ -4213,16 +4087,15 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
4213
4087
|
* on, named by step. A step that died of a full disk (docs/reference/specs/resident-repos.md item 54) also
|
|
4214
4088
|
* flips the resident `degraded(disk-full: …)` — not serviceable, so the
|
|
4215
4089
|
* next dispatch goes cold without attaching (the card names the disk, not
|
|
4216
|
-
* `/etc/gitconfig.lock`)
|
|
4217
|
-
*
|
|
4218
|
-
* is in flight
|
|
4219
|
-
private async attachFailed(err: unknown
|
|
4090
|
+
* `/etc/gitconfig.lock`); the recovery decision lives in the next refresh
|
|
4091
|
+
* instance's entry gate (the cron's, within one bucket) — an attach never
|
|
4092
|
+
* stops the container itself: it is in flight. */
|
|
4093
|
+
private async attachFailed(err: unknown): Promise<ThreadErr> {
|
|
4220
4094
|
if (!(err instanceof StepError)) return { error: `attach-failed: ${errMsg(err)}`, status: 500 };
|
|
4221
4095
|
const failure = await this.classifyFailure(err.step, err.message);
|
|
4222
4096
|
if (!failure.diskFull) return { error: `attach-failed at ${err.step}: ${err.message}`, status: 500 };
|
|
4223
4097
|
console.log(`attach: ${failure.reason}`);
|
|
4224
4098
|
await this.setResidentState("degraded", failure.reason);
|
|
4225
|
-
await this.armRefresh(resource, DISK_FULL_REARM_S);
|
|
4226
4099
|
return { error: `attach-failed: ${failure.reason}`, status: 500 };
|
|
4227
4100
|
}
|
|
4228
4101
|
|
|
@@ -4294,7 +4167,6 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
4294
4167
|
threadKey,
|
|
4295
4168
|
refHint,
|
|
4296
4169
|
wantSha,
|
|
4297
|
-
resource,
|
|
4298
4170
|
slug,
|
|
4299
4171
|
t0,
|
|
4300
4172
|
facts,
|
|
@@ -4316,7 +4188,6 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
4316
4188
|
threadKey: string;
|
|
4317
4189
|
refHint: string | null;
|
|
4318
4190
|
wantSha: string | null;
|
|
4319
|
-
resource: string;
|
|
4320
4191
|
slug: string;
|
|
4321
4192
|
t0: number;
|
|
4322
4193
|
facts: RepoFacts;
|
|
@@ -4325,7 +4196,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
4325
4196
|
mode: ReturnType<typeof planReadonlyAttach>;
|
|
4326
4197
|
rollback: () => Promise<void>;
|
|
4327
4198
|
}): Promise<AttachOk | ThreadErr> {
|
|
4328
|
-
const { threadKey, refHint, wantSha,
|
|
4199
|
+
const { threadKey, refHint, wantSha, slug, t0, facts, record, binding, mode, rollback } = input;
|
|
4329
4200
|
|
|
4330
4201
|
// Command-level token mint — before the lock so mint latency
|
|
4331
4202
|
// never holds the mutex, and failure never blocks the attach.
|
|
@@ -4414,7 +4285,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
4414
4285
|
if (err instanceof StepError && err.step === "unknown-ref") {
|
|
4415
4286
|
return { error: `unknown-ref: ${err.message}`, status: 400 };
|
|
4416
4287
|
}
|
|
4417
|
-
return this.attachFailed(err
|
|
4288
|
+
return this.attachFailed(err);
|
|
4418
4289
|
}
|
|
4419
4290
|
|
|
4420
4291
|
let deps: { deps: ThreadDepsMechanism; reconciled: boolean; depsKey?: string };
|
|
@@ -4449,7 +4320,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
4449
4320
|
const s = await this.getStatus();
|
|
4450
4321
|
return { error: errMsg(err), status: 503, state: s.state, reason: "mirror-busy" };
|
|
4451
4322
|
}
|
|
4452
|
-
return this.attachFailed(err
|
|
4323
|
+
return this.attachFailed(err);
|
|
4453
4324
|
}
|
|
4454
4325
|
|
|
4455
4326
|
// The prior write time and token expiry never survive an attach: a
|
|
@@ -4470,12 +4341,9 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
4470
4341
|
...(credentialsWrittenAt !== undefined ? { credentialsWrittenAt } : {}),
|
|
4471
4342
|
...(credentialTokenExpiresAtMs !== undefined ? { tokenExpiresAtMs: credentialTokenExpiresAtMs } : {}),
|
|
4472
4343
|
} satisfies ThreadBinding);
|
|
4473
|
-
|
|
4474
|
-
|
|
4475
|
-
|
|
4476
|
-
// Item 55: the tree is on disk now — measure it (deferred; the `du` stays
|
|
4477
|
-
// off this hot path) so the next admission projects from current parts.
|
|
4478
|
-
await this.scheduleDiskMeasure(resource);
|
|
4344
|
+
// Item 55: the tree is on disk now; the next refresh instance's `measure`
|
|
4345
|
+
// step counts it — the admission's free-space term is a live `df`, and the
|
|
4346
|
+
// per-part projection it reads from the sample moves only with a cycle.
|
|
4479
4347
|
|
|
4480
4348
|
return {
|
|
4481
4349
|
workspace: binding.worktreePath,
|
|
@@ -5489,9 +5357,8 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5489
5357
|
if (!(await this.evictBinding(current, activeNow, `detach`, "detach"))) {
|
|
5490
5358
|
return { released: false, reason: "re-attached during eviction — kept", user };
|
|
5491
5359
|
}
|
|
5492
|
-
// Item 55: the tree is gone
|
|
5493
|
-
//
|
|
5494
|
-
await this.scheduleDiskMeasure((await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "");
|
|
5360
|
+
// Item 55: the tree is gone; the gauge catches up at the next refresh
|
|
5361
|
+
// instance's `measure` step, and the admission's `df` sees the space now.
|
|
5495
5362
|
return { released: true, user };
|
|
5496
5363
|
}
|
|
5497
5364
|
|
|
@@ -5578,84 +5445,64 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5578
5445
|
/** A refresh cycle past its idle/reconcile gates (fetching, rebuilding, snapshotting). */
|
|
5579
5446
|
private refreshesInFlight = 0;
|
|
5580
5447
|
|
|
5581
|
-
/**
|
|
5582
|
-
*
|
|
5583
|
-
*
|
|
5584
|
-
*
|
|
5585
|
-
|
|
5586
|
-
|
|
5587
|
-
* two sweep chains can never be armed by construction. */
|
|
5588
|
-
private sweepInFlight = false;
|
|
5589
|
-
|
|
5590
|
-
async onWorktreeSweep(payload: string): Promise<{ evicted: string[]; kept: number }> {
|
|
5591
|
-
const resource = payload || ((await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "");
|
|
5448
|
+
/** The worktree inactivity sweep — every refresh instance's `sweep` step
|
|
5449
|
+
* (item 23) and the `sweep-now` debug op. Removes worktrees whose binding
|
|
5450
|
+
* is idle past the TTL, releases the user to the pool, and KEEPS the
|
|
5451
|
+
* binding record marked evicted. Never wakes a slept container just to
|
|
5452
|
+
* delete files a sleep already destroyed. */
|
|
5453
|
+
async sweepWorktrees(resource: string): Promise<{ evicted: string[]; kept: number }> {
|
|
5592
5454
|
const evicted: string[] = [];
|
|
5593
5455
|
let kept = 0;
|
|
5594
|
-
|
|
5595
|
-
|
|
5596
|
-
|
|
5597
|
-
|
|
5598
|
-
|
|
5599
|
-
|
|
5600
|
-
|
|
5601
|
-
|
|
5602
|
-
|
|
5603
|
-
|
|
5604
|
-
|
|
5605
|
-
|
|
5606
|
-
|
|
5607
|
-
|
|
5608
|
-
|
|
5609
|
-
|
|
5610
|
-
|
|
5611
|
-
|
|
5612
|
-
|
|
5613
|
-
|
|
5614
|
-
|
|
5615
|
-
|
|
5616
|
-
|
|
5617
|
-
|
|
5618
|
-
|
|
5619
|
-
|
|
5620
|
-
|
|
5621
|
-
|
|
5622
|
-
const busyNow = this.threadOpsInFlight.get(binding.threadKey) ?? 0;
|
|
5623
|
-
if (!cleanIdle || busyNow > 0) {
|
|
5624
|
-
kept++;
|
|
5625
|
-
continue;
|
|
5626
|
-
}
|
|
5627
|
-
}
|
|
5628
|
-
// Re-read the binding too: a re-attach that completed inside the
|
|
5629
|
-
// clean-check await bumped lastAttachAt and rebuilt the tree — evicting
|
|
5630
|
-
// from this loop's stale snapshot would rm the fresh tree.
|
|
5631
|
-
const current = await this.ctx.storage.get<ThreadBinding>(threadBindingKey(binding.threadKey));
|
|
5632
|
-
if (!current || current.evicted || current.lastAttachAt !== binding.lastAttachAt) {
|
|
5456
|
+
const record = await this.registry()
|
|
5457
|
+
.getRecord(resource)
|
|
5458
|
+
.catch(() => null);
|
|
5459
|
+
const ttlDays = record?.worktreeTtlDays ?? WORKTREE_TTL_DAYS_DEFAULT;
|
|
5460
|
+
const cutoff = systemClock() - ttlDays * 86_400_000;
|
|
5461
|
+
const all = await this.ctx.storage.list<ThreadBinding>({ prefix: THREAD_KEY_PREFIX });
|
|
5462
|
+
const active = await this.isRuntimeActive().catch(() => false);
|
|
5463
|
+
const idleCutoff = systemClock() - CLEAN_IDLE_RELEASE_S * 1000;
|
|
5464
|
+
for (const binding of all.values()) {
|
|
5465
|
+
if (binding.evicted || !binding.user) continue;
|
|
5466
|
+
const last = Date.parse(binding.lastAttachAt);
|
|
5467
|
+
if (last >= cutoff) {
|
|
5468
|
+
// Not past the TTL. Still release it if it has been idle for an hour,
|
|
5469
|
+
// nothing is running on it, and the tree is provably clean — the run
|
|
5470
|
+
// that used it is over and there is nothing to preserve. A slept
|
|
5471
|
+
// container has NO tree any more (sleep destroys the disk), so an
|
|
5472
|
+
// idle binding on an inactive runtime is releasable outright: there is
|
|
5473
|
+
// nothing left to protect, only a pool user to give back. (Keeping
|
|
5474
|
+
// them would leave idle bindings on a sleeping resident until the
|
|
5475
|
+
// 7-day TTL.)
|
|
5476
|
+
const busy = this.threadOpsInFlight.get(binding.threadKey) ?? 0;
|
|
5477
|
+
const cleanIdle =
|
|
5478
|
+
last < idleCutoff && busy === 0 && (!active || (await this.worktreeCleanliness(binding)).clean);
|
|
5479
|
+
// Re-read right before removal: the clean check awaited (the DO
|
|
5480
|
+
// yields), so an exec that arrived meanwhile would otherwise have
|
|
5481
|
+
// its tree removed under it — same guard as detachThread.
|
|
5482
|
+
const busyNow = this.threadOpsInFlight.get(binding.threadKey) ?? 0;
|
|
5483
|
+
if (!cleanIdle || busyNow > 0) {
|
|
5633
5484
|
kept++;
|
|
5634
5485
|
continue;
|
|
5635
5486
|
}
|
|
5636
|
-
// `active` is re-read per binding: the container can wake mid-sweep (an
|
|
5637
|
-
// attach), and an eviction decided on a stale "inactive" would skip the
|
|
5638
|
-
// rm and orphan a real tree.
|
|
5639
|
-
const activeNow = await this.isRuntimeActive().catch(() => false);
|
|
5640
|
-
if (
|
|
5641
|
-
await this.evictBinding(
|
|
5642
|
-
current,
|
|
5643
|
-
activeNow,
|
|
5644
|
-
`worktree-sweep ${resource}`,
|
|
5645
|
-
last >= cutoff ? "clean-idle" : "ttl",
|
|
5646
|
-
)
|
|
5647
|
-
)
|
|
5648
|
-
evicted.push(binding.threadKey);
|
|
5649
|
-
else kept++;
|
|
5650
5487
|
}
|
|
5651
|
-
|
|
5652
|
-
|
|
5653
|
-
|
|
5654
|
-
this.
|
|
5655
|
-
if (
|
|
5656
|
-
|
|
5488
|
+
// Re-read the binding too: a re-attach that completed inside the
|
|
5489
|
+
// clean-check await bumped lastAttachAt and rebuilt the tree — evicting
|
|
5490
|
+
// from this loop's stale snapshot would rm the fresh tree.
|
|
5491
|
+
const current = await this.ctx.storage.get<ThreadBinding>(threadBindingKey(binding.threadKey));
|
|
5492
|
+
if (!current || current.evicted || current.lastAttachAt !== binding.lastAttachAt) {
|
|
5493
|
+
kept++;
|
|
5494
|
+
continue;
|
|
5495
|
+
}
|
|
5496
|
+
// `active` is re-read per binding: the container can wake mid-sweep (an
|
|
5497
|
+
// attach), and an eviction decided on a stale "inactive" would skip the
|
|
5498
|
+
// rm and orphan a real tree.
|
|
5499
|
+
const activeNow = await this.isRuntimeActive().catch(() => false);
|
|
5500
|
+
if (
|
|
5501
|
+
await this.evictBinding(current, activeNow, `worktree-sweep ${resource}`, last >= cutoff ? "clean-idle" : "ttl")
|
|
5502
|
+
)
|
|
5503
|
+
evicted.push(binding.threadKey);
|
|
5504
|
+
else kept++;
|
|
5657
5505
|
}
|
|
5658
|
-
if (evicted.length > 0) await this.scheduleDiskMeasure(resource); // item 55
|
|
5659
5506
|
return { evicted, kept };
|
|
5660
5507
|
}
|
|
5661
5508
|
|
|
@@ -5870,9 +5717,9 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5870
5717
|
return { ok: true, lastAttachAt };
|
|
5871
5718
|
}
|
|
5872
5719
|
|
|
5873
|
-
/** Debug: run the sweep pass now (the exact
|
|
5720
|
+
/** Debug: run the sweep pass now (the exact function the `sweep` step runs). */
|
|
5874
5721
|
async debugSweepNow(): Promise<{ evicted: string[]; kept: number }> {
|
|
5875
|
-
return this.
|
|
5722
|
+
return this.sweepWorktrees((await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "");
|
|
5876
5723
|
}
|
|
5877
5724
|
|
|
5878
5725
|
// -- event-triggered reclamation ---------------------------------------------
|
|
@@ -5908,7 +5755,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5908
5755
|
/** Reclaim worktrees whose ref is FINISHED: the branch vanished from the
|
|
5909
5756
|
* mirror (the refresh cycle's `fetch --prune` just ran) or its PR was
|
|
5910
5757
|
* merged/closed. Runs inside the refresh cycle — a poll on the existing
|
|
5911
|
-
*
|
|
5758
|
+
* cadence, since the GitHub App has webhooks off — and via /debug
|
|
5912
5759
|
* reclaim-now. Never touches the default branch, a busy thread, or a dirty
|
|
5913
5760
|
* tree (reclaimDecision); every keep is named. The eviction itself is the
|
|
5914
5761
|
* sweep's `evictBinding` with the same re-read guards. */
|
|
@@ -6098,8 +5945,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
6098
5945
|
instance: null,
|
|
6099
5946
|
skipped: null,
|
|
6100
5947
|
};
|
|
6101
|
-
const [
|
|
6102
|
-
this.listSchedules(REFRESH_CALLBACK),
|
|
5948
|
+
const [provisionRun, provisionDeadline, bindings] = await Promise.all([
|
|
6103
5949
|
this.listSchedules(PROVISION_RUN_CALLBACK),
|
|
6104
5950
|
this.listSchedules(PROVISIONING_CALLBACK),
|
|
6105
5951
|
this.ctx.storage.list<ThreadBinding>({ prefix: THREAD_KEY_PREFIX }),
|
|
@@ -6144,8 +5990,8 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
6144
5990
|
checkoutBackupId: snap.checkout.id,
|
|
6145
5991
|
}
|
|
6146
5992
|
: null,
|
|
5993
|
+
// The provisioning schedules, the one timer a resident has (item 3).
|
|
6147
5994
|
schedules: {
|
|
6148
|
-
refresh: refresh.length,
|
|
6149
5995
|
provisionRun: provisionRun.length,
|
|
6150
5996
|
provisionDeadline: provisionDeadline.length,
|
|
6151
5997
|
},
|
|
@@ -6160,9 +6006,8 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
6160
6006
|
map.get(inFlightKey("refresh")) as Lease | undefined,
|
|
6161
6007
|
map.get(inFlightKey("hydration")) as Lease | undefined,
|
|
6162
6008
|
),
|
|
6163
|
-
// Item 7:
|
|
6164
|
-
//
|
|
6165
|
-
// it last reported, and the last bucket the cron skipped.
|
|
6009
|
+
// Item 7: the lifecycle row (`workflow`), the instance last created with
|
|
6010
|
+
// the step it last reported, and the last bucket the cron skipped.
|
|
6166
6011
|
lifecycle: lifecycleOf(map.get(LIFECYCLE_KEY)),
|
|
6167
6012
|
refresh: {
|
|
6168
6013
|
instance: refreshRow.instance
|
|
@@ -6183,32 +6028,15 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
6183
6028
|
|
|
6184
6029
|
// -- debug surface (admin-scoped via POST /debug; used by live validation) ---
|
|
6185
6030
|
|
|
6031
|
+
/** The pending schedule rows: provisioning's two, the one timer a resident
|
|
6032
|
+
* has — a settled resident answers both empty (lifecycle.test.ts holds that
|
|
6033
|
+
* nothing else is ever scheduled). */
|
|
6186
6034
|
async debugSchedules(): Promise<Record<string, unknown>> {
|
|
6187
|
-
const [
|
|
6188
|
-
this.listSchedules(REFRESH_CALLBACK),
|
|
6035
|
+
const [provisionRun, provisionDeadline] = await Promise.all([
|
|
6189
6036
|
this.listSchedules(PROVISION_RUN_CALLBACK),
|
|
6190
6037
|
this.listSchedules(PROVISIONING_CALLBACK),
|
|
6191
|
-
this.listSchedules(SWEEP_CALLBACK),
|
|
6192
6038
|
]);
|
|
6193
|
-
return {
|
|
6194
|
-
}
|
|
6195
|
-
|
|
6196
|
-
/** Kill the refresh chain (simulates a dead alarm chain for watchdog tests). */
|
|
6197
|
-
async debugKillRefresh(): Promise<{ killed: boolean; remaining: number }> {
|
|
6198
|
-
this.deleteSchedules(REFRESH_CALLBACK);
|
|
6199
|
-
return { killed: true, remaining: (await this.listSchedules(REFRESH_CALLBACK)).length };
|
|
6200
|
-
}
|
|
6201
|
-
|
|
6202
|
-
/** Pull the next refresh forward to ~1s from now. A resident on the Workflow
|
|
6203
|
-
* lifecycle has no chain to pull: its next cycle is the instance the next
|
|
6204
|
-
* cron firing creates (`run-watchdog` runs that pass on demand). */
|
|
6205
|
-
async debugRefreshNow(): Promise<{ scheduled: boolean; lifecycle: ResidentLifecycle }> {
|
|
6206
|
-
const lifecycle = await this.getLifecycle();
|
|
6207
|
-
if (lifecycle === "workflow") return { scheduled: false, lifecycle };
|
|
6208
|
-
const resource = (await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "";
|
|
6209
|
-
this.deleteSchedules(REFRESH_CALLBACK);
|
|
6210
|
-
await this.schedule(1, REFRESH_CALLBACK, resource);
|
|
6211
|
-
return { scheduled: true, lifecycle };
|
|
6039
|
+
return { provisionRun, provisionDeadline };
|
|
6212
6040
|
}
|
|
6213
6041
|
|
|
6214
6042
|
/** Fault injection for the watchdog's stuck-onboarding path: re-persist
|
|
@@ -6234,23 +6062,21 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
6234
6062
|
}
|
|
6235
6063
|
|
|
6236
6064
|
/** Fault injection for the watchdog's auto-rebuild path: persist
|
|
6237
|
-
* `down` with a rehydration-flavored reason
|
|
6238
|
-
*
|
|
6239
|
-
*
|
|
6240
|
-
* Test-only semantics; admin scope. */
|
|
6065
|
+
* `down` with a rehydration-flavored reason (what a real goDown does), so
|
|
6066
|
+
* repeated watchdog passes can strike it up to the auto-rebuild without
|
|
6067
|
+
* corrupting real R2 objects. Test-only semantics; admin scope. */
|
|
6241
6068
|
async debugForceDown(reason: string): Promise<ResidentStatus> {
|
|
6242
6069
|
await this.setResidentState("down", reason);
|
|
6243
|
-
this.deleteSchedules(REFRESH_CALLBACK);
|
|
6244
6070
|
return this.getStatus();
|
|
6245
6071
|
}
|
|
6246
6072
|
|
|
6247
6073
|
/** Rebuild: the down→onboarding escape hatch — discard the recorded
|
|
6248
6074
|
* snapshots (R2 objects included) and reprovision from scratch through the
|
|
6249
|
-
* ordinary
|
|
6075
|
+
* ordinary provisioning pipeline, reusing the registry record's command
|
|
6250
6076
|
* table/ref/budget. `dryRun` returns the same itemized plan WITHOUT
|
|
6251
6077
|
* executing: nothing deleted, no state change, schedules untouched.
|
|
6252
6078
|
* Refused while the engine owns the state (onboarding/refreshing/
|
|
6253
|
-
* restoring) — two
|
|
6079
|
+
* restoring) — two cycles must never race the same disk. */
|
|
6254
6080
|
async rebuild(
|
|
6255
6081
|
resource: string,
|
|
6256
6082
|
defaultRef: string,
|
|
@@ -6338,18 +6164,16 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
6338
6164
|
threadBindings: number;
|
|
6339
6165
|
}> {
|
|
6340
6166
|
const status = await this.getStatus();
|
|
6341
|
-
const [prov, run
|
|
6167
|
+
const [prov, run] = await Promise.all([
|
|
6342
6168
|
this.listSchedules(PROVISIONING_CALLBACK),
|
|
6343
6169
|
this.listSchedules(PROVISION_RUN_CALLBACK),
|
|
6344
|
-
this.listSchedules(REFRESH_CALLBACK),
|
|
6345
|
-
this.listSchedules(SWEEP_CALLBACK),
|
|
6346
6170
|
]);
|
|
6347
6171
|
const snap = await this.ctx.storage.get<SnapshotRecord>(SNAPSHOT_KEY);
|
|
6348
6172
|
const ids = snap ? [snap.mirror.id, snap.checkout.id] : [];
|
|
6349
6173
|
const bindings = await this.ctx.storage.list<ThreadBinding>({ prefix: THREAD_KEY_PREFIX });
|
|
6350
6174
|
return {
|
|
6351
6175
|
...status,
|
|
6352
|
-
schedules: prov.length + run.length
|
|
6176
|
+
schedules: prov.length + run.length,
|
|
6353
6177
|
snapshotBackupIds: ids,
|
|
6354
6178
|
backupObjects: await this.countBackupObjects(ids),
|
|
6355
6179
|
threadBindings: bindings.size,
|
|
@@ -6373,8 +6197,6 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
6373
6197
|
let backupObjectsDeleted = 0;
|
|
6374
6198
|
this.deleteSchedules(PROVISIONING_CALLBACK);
|
|
6375
6199
|
this.deleteSchedules(PROVISION_RUN_CALLBACK);
|
|
6376
|
-
this.deleteSchedules(REFRESH_CALLBACK);
|
|
6377
|
-
this.deleteSchedules(SWEEP_CALLBACK);
|
|
6378
6200
|
const snap = await this.ctx.storage.get<SnapshotRecord>(SNAPSHOT_KEY);
|
|
6379
6201
|
if (snap) {
|
|
6380
6202
|
try {
|
|
@@ -6751,11 +6573,13 @@ export default {
|
|
|
6751
6573
|
return res;
|
|
6752
6574
|
},
|
|
6753
6575
|
|
|
6754
|
-
/** Watchdog cron: one sparse pass that
|
|
6755
|
-
* (
|
|
6756
|
-
* invariant: this cron (every 10
|
|
6757
|
-
*
|
|
6758
|
-
* the
|
|
6576
|
+
/** Watchdog cron: one sparse pass that creates each resident's due refresh
|
|
6577
|
+
* instance (item 7), names a stale mid-flight marker by its instance, and
|
|
6578
|
+
* times out stuck onboarding. Cadence invariant: this cron (every 10
|
|
6579
|
+
* minutes) stays SHORTER than SLEEP_AFTER ("20m") — the instance it
|
|
6580
|
+
* creates is the keep-warm. It reads DO storage and the engine's instance
|
|
6581
|
+
* status only — containers are started by the instances' steps, not by the
|
|
6582
|
+
* watchdog itself.
|
|
6759
6583
|
*
|
|
6760
6584
|
* The cron is the `resident` entry of the schedule registry
|
|
6761
6585
|
* (src/core/schedules.ts — a unit test keeps wrangler.jsonc equal to it);
|
|
@@ -7102,7 +6926,7 @@ async function handleReconfigure(env: Env, body: Record<string, unknown>): Promi
|
|
|
7102
6926
|
* reusing the registry record (command table, ref, budget) as-is; the cap
|
|
7103
6927
|
* slot and thread bindings are untouched. `dryRun:true` answers 200 with the
|
|
7104
6928
|
* itemized plan and executes nothing; a real rebuild answers 202 like
|
|
7105
|
-
* onboard (the transition is
|
|
6929
|
+
* onboard (the transition is provisioning's schedule). */
|
|
7106
6930
|
async function handleRebuild(env: Env, body: Record<string, unknown>): Promise<Response> {
|
|
7107
6931
|
const resource = parseResource(body.resource);
|
|
7108
6932
|
if ("error" in resource) return json({ error: resource.error }, 400);
|
|
@@ -7464,8 +7288,8 @@ function streamOp(pending: Promise<Awaited<ReturnType<ResidentDO["runOp"]>>>): R
|
|
|
7464
7288
|
);
|
|
7465
7289
|
}
|
|
7466
7290
|
|
|
7467
|
-
/** Admin diagnostic surface, used by the live validation of the freshness engine (
|
|
7468
|
-
* stop-container
|
|
7291
|
+
/** Admin diagnostic surface, used by the live validation of the freshness engine (refresh-now
|
|
7292
|
+
* creates this bucket's instance on demand, stop-container simulates a platform sleep; mint-token proves
|
|
7469
7293
|
* the command-level mint failure shape without exposing token material).
|
|
7470
7294
|
* Side-effect-explicit; every op is admin-scope except the pure reads
|
|
7471
7295
|
* info/schedules/threads, which the read scope may also run. */
|
|
@@ -7513,10 +7337,14 @@ async function handleDebug(env: Env, body: Record<string, unknown>): Promise<Res
|
|
|
7513
7337
|
return json(await stub.getResidentInfo());
|
|
7514
7338
|
case "schedules":
|
|
7515
7339
|
return json(await stub.debugSchedules());
|
|
7516
|
-
case "kill-refresh":
|
|
7517
|
-
return json(await stub.debugKillRefresh());
|
|
7518
7340
|
case "refresh-now":
|
|
7519
|
-
|
|
7341
|
+
// Item 13: this bucket's refresh instance, created now — `duplicate` when
|
|
7342
|
+
// the cron already served the bucket, `skipped` beside a live cycle.
|
|
7343
|
+
return json({
|
|
7344
|
+
op,
|
|
7345
|
+
resource: resource.resource,
|
|
7346
|
+
...(await createRefreshInstanceNow(env, stub, resource.resource)),
|
|
7347
|
+
});
|
|
7520
7348
|
case "stop-container":
|
|
7521
7349
|
return json(await stub.debugStopContainer());
|
|
7522
7350
|
case "force-onboarding":
|
|
@@ -7555,16 +7383,16 @@ async function handleDebug(env: Env, body: Record<string, unknown>): Promise<Res
|
|
|
7555
7383
|
return json(await stub.debugBackdateThread(threadKey.threadKey, days.value));
|
|
7556
7384
|
}
|
|
7557
7385
|
case "lifecycle": {
|
|
7558
|
-
// Item 7:
|
|
7559
|
-
//
|
|
7386
|
+
// Item 7: the lifecycle row. `workflow` is the one scheduler, so the op
|
|
7387
|
+
// only rewrites a stale `alarm` value the flagged rollout left behind.
|
|
7560
7388
|
const mode = parseLifecycle(body.mode);
|
|
7561
|
-
if (!mode) return json({ error: 'mode must be "
|
|
7389
|
+
if (!mode) return json({ error: 'mode must be "workflow" — the alarm chain no longer exists' }, 400);
|
|
7562
7390
|
return json({ op, resource: resource.resource, ...(await stub.setLifecycle(mode)) });
|
|
7563
7391
|
}
|
|
7564
7392
|
default:
|
|
7565
7393
|
return json(
|
|
7566
7394
|
{
|
|
7567
|
-
error: `unknown op ${JSON.stringify(op)} (ops: info, schedules,
|
|
7395
|
+
error: `unknown op ${JSON.stringify(op)} (ops: info, schedules, refresh-now, stop-container, force-onboarding, force-down, mint-token, run-watchdog, set-test-overrides, threads, sweep-now, reclaim-now, measure-disk, purge-bindings, backdate-thread, lifecycle)`,
|
|
7568
7396
|
},
|
|
7569
7397
|
400,
|
|
7570
7398
|
);
|
|
@@ -7575,8 +7403,8 @@ async function handleDebug(env: Env, body: Record<string, unknown>): Promise<Res
|
|
|
7575
7403
|
* handler and the /debug run-watchdog op. Each check targets a different DO,
|
|
7576
7404
|
* so they run concurrently; a failing one becomes its own {error} entry
|
|
7577
7405
|
* without touching its neighbors, and the results follow the registry list.
|
|
7578
|
-
*
|
|
7579
|
-
*
|
|
7406
|
+
* The pass also creates each resident's refresh instance when its bucket is
|
|
7407
|
+
* due (item 7). */
|
|
7580
7408
|
async function runWatchdog(env: Env, parent?: TraceSpan): Promise<WatchdogSummary> {
|
|
7581
7409
|
const registry = registryStub(env);
|
|
7582
7410
|
const residents = await registry.list();
|
|
@@ -7608,7 +7436,6 @@ async function runWatchdog(env: Env, parent?: TraceSpan): Promise<WatchdogSummar
|
|
|
7608
7436
|
reason: s.value.reason,
|
|
7609
7437
|
action: s.value.action,
|
|
7610
7438
|
disk: s.value.disk,
|
|
7611
|
-
lifecycle: s.value.refresh.lifecycle,
|
|
7612
7439
|
instance: s.value.instance,
|
|
7613
7440
|
}
|
|
7614
7441
|
: { resource: record.resource, error: errMsg(s.reason) };
|