@coreplane/switchboard 1.202.1 → 1.204.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/dist/assets/config/config.example.yaml +10 -0
  2. package/dist/assets/deploy/cloudflare-resident/preflight.mjs +3 -3
  3. package/dist/assets/deploy/cloudflare-resident/refresh.ts +179 -93
  4. package/dist/assets/deploy/cloudflare-resident/shared.ts +15 -10
  5. package/dist/assets/deploy/cloudflare-resident/worker.ts +459 -632
  6. package/dist/assets/package-lock.json +3 -3
  7. package/dist/assets/package.json +1 -1
  8. package/dist/assets/project.json +1 -1
  9. package/dist/assets/source.json +3 -3
  10. package/dist/assets/src/core/authz/policy.ts +4 -0
  11. package/dist/assets/src/core/runRecord.ts +11 -0
  12. package/dist/assets/src/core/schedules.ts +12 -6
  13. package/dist/assets/src/core/ship/handoff.ts +246 -0
  14. package/dist/assets/src/execution/residentDisk.ts +2 -2
  15. package/dist/assets/src/execution/residentInstanceId.ts +48 -45
  16. package/dist/assets/src/execution/residentRefresh.ts +17 -66
  17. package/dist/assets/src/execution/residentState.ts +8 -9
  18. package/dist/assets/src/execution/residentStepPlan.ts +1 -1
  19. package/dist/assets/web/dist/.vite/manifest.json +55 -44
  20. package/dist/assets/web/dist/assets/{AppShell-CJLPO_aI.js → AppShell-Bw3-_TTE.js} +1 -1
  21. package/dist/assets/web/dist/assets/{CostsPage-C6ZvnAmH.js → CostsPage-XQ-fQzdz.js} +1 -1
  22. package/dist/assets/web/dist/assets/DeliveryPage-BRwcQyr7.js +1 -0
  23. package/dist/assets/web/dist/assets/{NotFoundPage-BdLcY60r.js → NotFoundPage-RjH-9dPy.js} +1 -1
  24. package/dist/assets/web/dist/assets/{ResidentDetailPage-hXWrr8T6.js → ResidentDetailPage-BRy5wkv9.js} +1 -1
  25. package/dist/assets/web/dist/assets/{ResidentsIndexPage-YLu4JwxJ.js → ResidentsIndexPage-DAEVLN8j.js} +1 -1
  26. package/dist/assets/web/dist/assets/{RunRoutePage-BDhHRJVO.js → RunRoutePage-B1KHmkZ9.js} +1 -1
  27. package/dist/assets/web/dist/assets/{RunsIndexPage-D7zZ5a0l.js → RunsIndexPage-3hUWFpFV.js} +1 -1
  28. package/dist/assets/web/dist/assets/{RunsTabs-BOUlSa2W.js → RunsTabs-Qn_5TVJU.js} +1 -1
  29. package/dist/assets/web/dist/assets/{ScheduledPage-BciOQcC-.js → ScheduledPage-DysVvVm1.js} +1 -1
  30. package/dist/assets/web/dist/assets/{StatusDot-CTDLBv92.js → StatusDot-Bug2a6T6.js} +1 -1
  31. package/dist/assets/web/dist/assets/{Tooltip-DD2v9Gxx.js → Tooltip-C2eEUbwn.js} +1 -1
  32. package/dist/assets/web/dist/assets/{favicon-C4Q8cRsP.js → favicon-CFfbl2AI.js} +1 -1
  33. package/dist/assets/web/dist/assets/{main-DrSlUcMg.js → main-0uhp-taL.js} +3 -3
  34. package/dist/assets/web/dist/assets/main-Xickfv8V.css +1 -0
  35. package/dist/cli.js +3065 -1912
  36. package/package.json +1 -1
  37. package/dist/assets/web/dist/assets/main-DoTjwE-G.css +0 -1
@@ -17,12 +17,15 @@
17
17
  // operator scope POST /attach /detach /exec /read /write /op GET /status (state, reason, inFlight)
18
18
  // unauthenticated GET /healthz (deploy wake ping; touches no DO)
19
19
  //
20
- // Lifecycle engine: alarm-driven provisioning (clone → install/build →
21
- // stamped snapshot → warm), wake-path rehydration (`restoring` persisted
22
- // BEFORE restore, stamped snapshots refused on mismatch), a self-rescheduling
23
- // refresh alarm, and a cron watchdog (re-arm dead chains + degraded
24
- // (alarm-missed); time out stuck onboarding; auto-rebuild after N consecutive
25
- // down passes on a rehydration-flavored reason). GitHub App tokens are minted
20
+ // Lifecycle engine: schedule-driven provisioning (clone → install/build →
21
+ // stamped snapshot → warm; the one timer left), wake-path rehydration
22
+ // (`restoring` persisted BEFORE restore, stamped snapshots refused on
23
+ // mismatch), the refresh cycle as a Workflow instance the cron creates per
24
+ // resident and ten-minute bucket (refresh.ts: fetch, install, build,
25
+ // snapshot, then the worktree sweep and the disk measurement as steps), and
26
+ // a cron watchdog (create the due instances; name a stale mid-flight marker
27
+ // by its instance; time out stuck onboarding; auto-rebuild after N
28
+ // consecutive down passes on a rehydration-flavored reason). GitHub App tokens are minted
26
29
  // repo-scoped on WebCrypto. POST /rebuild is the down→onboarding escape hatch
27
30
  // (discard snapshots, reprovision from scratch); /offboard and /rebuild
28
31
  // support dryRun (itemized plan, nothing executed); onboard verifies GitHub
@@ -131,7 +134,6 @@ import {
131
134
  classifyRefreshFailure,
132
135
  restoreFailureDisposition,
133
136
  killStaleBuildProcessesCommand,
134
- nextRefreshDelayS,
135
137
  planRefresh,
136
138
  RUNTIME_REPLACEMENT_WORDING,
137
139
  judgeRestoreProgress,
@@ -144,7 +146,6 @@ import {
144
146
  type RefreshPlan,
145
147
  type RestoreSample,
146
148
  type RefreshFailure,
147
- type RefreshOutcome,
148
149
  } from "../../src/execution/residentRefresh.js";
149
150
  import {
150
151
  lifecycleOf,
@@ -250,13 +251,13 @@ import {
250
251
  planDepsMaterialization,
251
252
  } from "../../src/execution/residentDepsStore.js";
252
253
  import { buildId, injectedBuildStamp } from "../../src/deploy/buildStamp.js";
253
- import { createRefreshInstance, type RefreshInstanceParams } from "./refresh";
254
+ import { createRefreshInstance, createRefreshInstanceNow, type RefreshInstanceParams } from "./refresh";
254
255
  import {
255
256
  DEFAULT_EXEC_TIMEOUT_MS,
256
257
  DEPS_STEP_OVERHEAD_MS,
257
258
  errMsg,
258
259
  GIT_NETWORK_TIMEOUT_MS,
259
- IDLE_REFRESH_INTERVAL_S,
260
+ THREAD_POOL_SIZE,
260
261
  R2_TRANSFER_TIMEOUT_MS,
261
262
  REFRESH_BUILD_TIMEOUT_MS,
262
263
  REFRESH_INSTALL_TIMEOUT_MS,
@@ -317,8 +318,7 @@ export interface Env {
317
318
  BACKUP_BUCKET: R2Bucket;
318
319
  /** The refresh cycle as a Workflow instance (docs/reference/specs/resident-repos.md
319
320
  * item 7): `ResidentRefresh` in refresh.ts, re-exported above. The watchdog
320
- * cron creates one per resident whose row says `lifecycle: workflow`; an
321
- * `alarm` resident (the default) never has one. */
321
+ * cron creates one per resident and ten-minute bucket. */
322
322
  RESIDENT_REFRESH: Workflow<RefreshInstanceParams>;
323
323
  // Presigned snapshot transfers (docs/reference/specs/resident-repos.md item 61): with all
324
324
  // four present the container moves archive bytes itself over presigned R2
@@ -436,7 +436,7 @@ const OPS_DIR = "/workspace/ops";
436
436
  * repo are genuinely concurrent. Memory, not this list, is the real ceiling
437
437
  * — see the instance_type note in wrangler.jsonc. Must match the useradd loop
438
438
  * in the Dockerfile. */
439
- const THREAD_USERS = Array.from({ length: 16 }, (_, i) => `worker${i + 2}`);
439
+ const THREAD_USERS = Array.from({ length: THREAD_POOL_SIZE }, (_, i) => `worker${i + 2}`);
440
440
 
441
441
  /** Force-detach: after killing the thread user's processes, how long
442
442
  * to wait for the in-flight op counter to drain (polled every
@@ -457,21 +457,16 @@ const FORCE_DETACH_KILL_TIMEOUT_MS = 2_000;
457
457
  * binding record is KEPT so the next attach recreates with the same
458
458
  * ref. Overridable per resident via the onboard-time `worktreeTtlDays`. */
459
459
  const WORKTREE_TTL_DAYS_DEFAULT = 7;
460
- /** The sweep self-reschedules hourly (armed by attach when no sweep pends). */
461
- const SWEEP_INTERVAL_S = 60 * 60; // hourly: the sweep is the backstop for trees a run kept (dirty) or never released
462
460
  /** A live binding whose last attach is older than this AND whose tree is clean
463
- * (no uncommitted/unpushed work) is released by the hourly sweep — runs that
464
- * ended before /detach existed, or whose release call was lost. Dirty trees
465
- * keep to the TTL. */
461
+ * (no uncommitted/unpushed work) is released by the sweep step (every refresh
462
+ * instance runs one) — runs that ended before /detach existed, or whose
463
+ * release call was lost. Dirty trees keep to the TTL. */
466
464
  const CLEAN_IDLE_RELEASE_S = 60 * 60;
467
- /** Slack over SWEEP_INTERVAL_S before a pending sweep row counts as config
468
- * drift (armed by older code with a longer interval). A healthy row is due at
469
- * most SWEEP_INTERVAL_S out and only gets closer, so this never trips on one. */
470
- const SWEEP_DRIFT_SLACK_S = 5 * 60;
471
465
  /** Idle sleep: when no thread has attached within this window and no live
472
- * tree is dirty, the refresh alarm skips the fetch and re-arms far out so
473
- * the container can actually sleep (SLEEP_AFTER); the next attach refreshes
474
- * first if the mirror is stale (refresh-on-attach). */
466
+ * tree is dirty, the refresh cycle skips the fetch and the cron holds the
467
+ * next instance to the idle cadence, so the container can actually sleep
468
+ * (SLEEP_AFTER); the next attach refreshes first if the mirror is stale
469
+ * (refresh-on-attach). */
475
470
  const IDLE_AFTER_S = 60 * 60;
476
471
  /** LRU eviction floor: an over-cap onboard with `evictColdest:true` may
477
472
  * offboard the coldest eligible warm resident, but never one whose last
@@ -487,49 +482,42 @@ const GITHUB_API_TIMEOUT_MS = 10_000;
487
482
  * idle-park like a warm one; the next attach still refreshes first. */
488
483
  const DEGRADED_PARK_AFTER_CYCLES = 3;
489
484
  const DEGRADED_STREAK_KEY = "resident:degradedStreak";
490
- /** Degraded reasons stamped by the WATCHDOG rather than by an attempted refresh
491
- * (`watchdogCheck`: `alarm-missed: …`, `stale-mid-flight: …` — both always
492
- * carry a `: detail` suffix). They mean "a cycle must run", so they never
493
- * count toward the park streak. Deliberate trade-off: a resident that
494
- * oscillates between a refresh-produced failure and watchdog stamps (e.g.
495
- * `install-failed` → DO eviction → `stale-mid-flight` → `install-failed` …)
496
- * keeps resetting the streak and never parks — full 10-min cadence for a
497
- * chronically broken repo. Accepted: a watchdog stamp means the previous
498
- * "same reason" observation is not trustworthy, and preserving the streak
499
- * across it would re-open the parked-degraded hole this fixes. */
500
- /** Plus a cycle whose step was killed from OUTSIDE by a deploy
501
- * (`refresh-interrupted: …`, classified by `classifyRefreshFailure`): equally
502
- * not evidence about the repository, equally never counted. */
503
- const NON_EVIDENCE_REASON = /^(?:alarm-missed|stale-mid-flight|refresh-interrupted|restore-interrupted):/;
504
- /** Consecutive cycles that ended `refresh-interrupted`: feeds the
505
- * short-re-arm cap in `nextRefreshDelayS`; cleared by any other outcome. */
506
- const INTERRUPTED_STREAK_KEY = "resident:interruptedStreak";
485
+ /** Degraded reasons that are not evidence about the repository — both always
486
+ * carry a `: detail` suffix: the watchdog's `stale-mid-flight: …` (a marker
487
+ * a dead cycle left behind; a cycle must run) and the wake path's
488
+ * `restore-interrupted: …` (the runtime was replaced under a restore; the
489
+ * step's retry restores again). They never count toward the park streak.
490
+ * Deliberate trade-off: a resident that oscillates between a
491
+ * refresh-produced failure and a watchdog stamp (e.g. `install-failed` → DO
492
+ * eviction → `stale-mid-flight` → `install-failed` …) keeps resetting the
493
+ * streak and never parks — full 10-min cadence for a chronically broken
494
+ * repo. Accepted: a watchdog stamp means the previous "same reason"
495
+ * observation is not trustworthy, and preserving the streak across it would
496
+ * re-open the parked-degraded hole this fixes. A refresh step killed from
497
+ * outside (`refresh-interrupted`, `classifyRefreshFailure`) is never recorded
498
+ * as `degraded` at all: the instance throws it to the engine, whose retry
499
+ * re-enters the step. */
500
+ const NON_EVIDENCE_REASON = /^(?:stale-mid-flight|restore-interrupted):/;
507
501
  /** When the disk-full recovery last stopped the container (docs/reference/specs/resident-repos.md item 54):
508
502
  * feeds `planDiskFullRecovery`'s cooldown so a working set that refills the
509
503
  * disk is named, not recycled in a loop. */
510
504
  const DISK_FULL_RECYCLE_KEY = "resident:diskFullRecycleAt";
511
- /** A disk-full attach pulls the refresh cycle this close (seconds) so the
512
- * recovery decision runs now, not at the next 600 s alarm. */
513
- const DISK_FULL_REARM_S = 1;
514
505
  /** The last disk measurement (docs/reference/specs/resident-repos.md item 55; `residentDiskBudget.ts`): one
515
- * `df` + one `du` over the parts, taken at the end of every refresh cycle and
516
- * (deferred by DISK_MEASURE_DELAY_S, off the hot path) after every attach,
517
- * detach and sweep eviction. Surfaced as the live view's `disk`; the attach
518
- * admission projects a new tree's cost from its parts. */
506
+ * `df` + one `du` over the parts, taken by every refresh instance's `measure`
507
+ * step after its sweep. Surfaced as the live view's `disk`; the attach
508
+ * admission projects a new tree's cost from its parts (its free-space term
509
+ * is a live `df` of its own). */
519
510
  const DISK_KEY = "resident:disk";
520
- /** Which scheduler drives this resident's refresh cycle (docs/reference/specs/resident-repos.md
521
- * item 7): the alarm chain (the default — a row without the key reads `alarm`)
522
- * or the Workflow instance the watchdog cron creates. Set through the admin
523
- * `/debug` `lifecycle` op; read by the alarm's entry, the watchdog's re-arm
524
- * branches and the cron's instance-creation duty, so the two schedulers never
525
- * both drive a cycle for one resident. */
511
+ /** The lifecycle row (docs/reference/specs/resident-repos.md item 7): `workflow`,
512
+ * the one scheduler. Kept from the flagged rollout so `/status` and `/debug
513
+ * info` can say so; an `alarm` value a flip left behind reads `workflow`
514
+ * (`lifecycleOf`) — the chain it named no longer exists. Rewritten by the
515
+ * admin `/debug` `lifecycle` op. */
526
516
  const LIFECYCLE_KEY = "resident:lifecycle";
527
- /** The refresh instance row (item 7): the last instance the cron created for
528
- * this resident, with the step it last reported and the cycle lease it holds,
529
- * and the last bucket the cron skipped (a live cycle, a duplicate id). */
517
+ /** The refresh instance row (item 7): the last instance created for this
518
+ * resident, with the step it last reported and the cycle lease it holds, and
519
+ * the last bucket the cron skipped (a live cycle, a duplicate id). */
530
520
  const REFRESH_INSTANCE_KEY = "resident:refreshInstance";
531
- const DISK_MEASURE_CALLBACK = "onDiskMeasure";
532
- const DISK_MEASURE_DELAY_S = 1;
533
521
  /** A `du` over a multi-GB checkout plus every live tree is seconds warm, tens
534
522
  * of seconds on a cold page cache — the same class as a git network step. */
535
523
  const DU_TIMEOUT_MS = GIT_NETWORK_TIMEOUT_MS;
@@ -1030,7 +1018,7 @@ interface RepoFacts {
1030
1018
  lastRefreshAt: string;
1031
1019
  lastRefreshError?: string; // last cycle's failure reason: command-level (e.g. token mint, no lifecycle flip) or the classified reason of a failed/interrupted cycle (survives a concurrent state overwrite); cleared by the next completed cycle
1032
1020
  lastRestore?: { at: string; ms: number }; // proof of restore-not-reclone on the wake path
1033
- /** Set while the resident is in idle mode (refresh alarm parked far out so the container may sleep). */
1021
+ /** Set while the resident is in idle mode (the cron holds the next refresh instance to the idle cadence so the container may sleep). */
1034
1022
  idleSince?: string;
1035
1023
  }
1036
1024
 
@@ -1083,6 +1071,17 @@ interface RefreshInstanceRow {
1083
1071
  skipped: { id: string; at: string; why: string } | null;
1084
1072
  }
1085
1073
 
1074
+ /** The engine's statuses under which an instance is still a live cycle:
1075
+ * queued, running, paused or waiting. Anything else — `complete`, `errored`,
1076
+ * `terminated`, an id the engine does not know — is not. */
1077
+ const INSTANCE_LIVE_STATUSES: ReadonlySet<string> = new Set([
1078
+ "queued",
1079
+ "running",
1080
+ "paused",
1081
+ "waiting",
1082
+ "waitingForPause",
1083
+ ]);
1084
+
1086
1085
  /** What every instance step answers besides its own facts: the resident's
1087
1086
  * wall clock at the step's start and the commands it ran, so the instance
1088
1087
  * can graft them under its root the way the bot grafts an attach's. */
@@ -1242,8 +1241,16 @@ export class ResidentRegistryDO extends DurableObject<Env> {
1242
1241
 
1243
1242
  const PROVISIONING_CALLBACK = "onProvisioningDeadline"; // fail-closed deadline
1244
1243
  const PROVISION_RUN_CALLBACK = "runProvisioning"; // the actual provisioning work
1245
- const REFRESH_CALLBACK = "onRefreshAlarm"; // self-rescheduling freshness chain
1246
- const SWEEP_CALLBACK = "onWorktreeSweep"; // hourly worktree inactivity eviction
1244
+ /** The schedule callbacks the retired alarm chain armed, gone with the flip to
1245
+ * the Workflow scheduler. Their rows outlive the code that armed them, and the
1246
+ * SDK's `alarm()` skips a due row whose callback method is gone WITHOUT
1247
+ * deleting it, then re-arms for that past time at once — a hot alarm loop
1248
+ * that only the row's deletion ends (@cloudflare/containers 0.3.7,
1249
+ * dist/lib/container.js `alarm()`: the `continue` precedes the DELETE). The
1250
+ * constructor deletes them before the first event, which on an upgraded
1251
+ * resident is that very alarm. Drop this list once every resident has woken on
1252
+ * this code (lifecycle.test.ts pins the names). */
1253
+ const RETIRED_SCHEDULE_CALLBACKS = ["onRefreshAlarm", "onWorktreeSweep", "onDiskMeasure"] as const;
1247
1254
 
1248
1255
  const STATE_KEY = "resident:state";
1249
1256
  const REASON_KEY = "resident:reason";
@@ -1298,20 +1305,44 @@ class StepError extends Error {
1298
1305
  }
1299
1306
 
1300
1307
  /** Thrown after the resident has ALREADY been transitioned to `down` (reason
1301
- * persisted); signals callers to stop the refresh chain without re-flipping. */
1308
+ * persisted); signals callers the cycle is over without re-flipping. */
1302
1309
  class ResidentDownError extends Error {
1303
1310
  constructor(public reason: string) {
1304
1311
  super(reason);
1305
1312
  }
1306
1313
  }
1307
1314
 
1315
+ /** Thrown by a refresh gate that stopped the container on purpose — a stale
1316
+ * image (`reconcileImage`), a disk-full recycle (`recoverFromDiskFull`) — so
1317
+ * the instance step it runs in throws to the engine, whose retry (thirty
1318
+ * seconds on) finds the container back on the current image or an empty
1319
+ * disk, restores it and runs the cycle: the retry is the re-warm that a
1320
+ * short re-arm used to be. Not a failure of the repository's own — never
1321
+ * recorded as `degraded`. */
1322
+ class CycleRestartError extends Error {
1323
+ constructor(public why: "image-stale-restart" | "disk-full-restart") {
1324
+ super(`${why}: the container is restarting — the step is retried onto it`);
1325
+ }
1326
+ }
1327
+
1308
1328
  export class ResidentDO extends Sandbox<Env> {
1309
- // TIMER RULE: never call ctx.storage.setAlarm/deleteAlarm from lifecycle
1310
- // code — the Container base class owns the DO alarm slot (its sleepAfter
1311
- // machinery and schedule multiplexing live there). All resident timers go
1312
- // through this.schedule()/this.deleteSchedules(), which multiplex onto that
1313
- // alarm safely. (Checked against @cloudflare/containers 0.3.7: the SDK
1314
- // registers no schedule callback names, so ours cannot collide.)
1329
+ // TIMER RULE: lifecycle code never arms the Durable Object's own alarm slot
1330
+ // — the Container base class owns it (its sleepAfter machinery and the
1331
+ // schedule multiplexing live there). The one timer left is provisioning's
1332
+ // (`initResident`: the run at +1 s and its fail-closed deadline), through
1333
+ // the base class's schedule API, which multiplexes onto that slot safely
1334
+ // (checked against @cloudflare/containers 0.3.7: the SDK registers no
1335
+ // schedule callback names, so ours cannot collide). Every other cycle is a
1336
+ // Workflow instance (refresh.ts) and the watchdog re-arms nothing;
1337
+ // lifecycle.test.ts holds the line over these sources.
1338
+
1339
+ constructor(...args: ConstructorParameters<typeof Sandbox<Env>>) {
1340
+ super(...args);
1341
+ // The base class created `container_schedules` synchronously above; the
1342
+ // rows the retired alarm chain armed are gone before this object handles
1343
+ // its first event (RETIRED_SCHEDULE_CALLBACKS).
1344
+ for (const name of RETIRED_SCHEDULE_CALLBACKS) this.deleteSchedules(name);
1345
+ }
1315
1346
 
1316
1347
  /** Serializes concurrent hydration attempts within one DO lifetime. Never
1317
1348
  * used as a "hydrated" flag — the container can sleep while the DO object
@@ -1978,8 +2009,8 @@ export class ResidentDO extends Sandbox<Env> {
1978
2009
  // about to change and asks the pure plan (residentStepPlan.ts) whether the
1979
2010
  // work is done — done issues no command, so a second call with the same
1980
2011
  // inputs has no effect — then takes its lease, runs its commands under the
1981
- // step's own budget, writes its result and releases. The alarm chain drives
1982
- // them today in the order it always did.
2012
+ // step's own budget, writes its result and releases. The refresh instance
2013
+ // drives them one step at a time (`refreshInstance*`).
1983
2014
 
1984
2015
  /** Fetch the mirror from origin, once per cycle: the record under
1985
2016
  * LAST_FETCH_KEY names the cycle, so a repeated call inside the same cycle
@@ -2205,9 +2236,9 @@ export class ResidentDO extends Sandbox<Env> {
2205
2236
  * or capped restore leaves the resident `down` with the reason and the
2206
2237
  * container stopped, as the wake path always did; a restore the runtime
2207
2238
  * replacement interrupts (a deploy rolled the container under it) is
2208
- * `restore-interrupted`, degraded and rethrown for the cycle to re-arm
2209
- * short — nothing is streaming into a disk that no longer exists
2210
- * (restoreFailureDisposition). */
2239
+ * `restore-interrupted`, degraded and rethrown for the instance step that
2240
+ * called it to retry — nothing is streaming into a disk that no longer
2241
+ * exists (restoreFailureDisposition). */
2211
2242
  async restoreCheckout(snap: SnapshotRecord, deadlineMs: number): Promise<{ done: boolean }> {
2212
2243
  const plan = planRestore({ sha: snap.sha, readyStamp: await this.readyStamp() });
2213
2244
  if (plan.action === "done") return { done: true };
@@ -2265,9 +2296,9 @@ export class ResidentDO extends Sandbox<Env> {
2265
2296
  // container, so nothing can land on a rebuild and there is nothing to
2266
2297
  // stop. Not evidence about the repo — the resident is `degraded` with
2267
2298
  // the restore named, never `down`, and the error goes back to the
2268
- // cycle, whose classifier reads the same wording as an interruption
2269
- // and re-arms short; the next wake restores again onto the new
2270
- // container. Before this branch every such restore ended `down`, and
2299
+ // instance step, whose classifier reads the same wording as an
2300
+ // interruption and throws to the engine; the retry restores again onto
2301
+ // the new container. Before this branch every such restore ended `down`, and
2271
2302
  // only a rebuild (the watchdog's, after three passes) brought the
2272
2303
  // resident back.
2273
2304
  this.swapIncarnation(); // the container this incarnation's memos described is gone
@@ -2325,16 +2356,11 @@ export class ResidentDO extends Sandbox<Env> {
2325
2356
  return counts.reduce((a, n) => a + n, 0);
2326
2357
  }
2327
2358
 
2328
- private async armRefresh(resource: string, intervalS = REFRESH_INTERVAL_S): Promise<void> {
2329
- this.deleteSchedules(REFRESH_CALLBACK); // at most one pending refresh
2330
- await this.schedule(intervalS, REFRESH_CALLBACK, resource);
2331
- }
2332
-
2333
- /** Persist `down` with a reason, stop the refresh chain, and hand back the
2334
- * error that tells callers the transition already happened. */
2359
+ /** Persist `down` with a reason and hand back the error that tells callers
2360
+ * the transition already happened (a down resident gets no refresh
2361
+ * instance: the cron's decision reads the state). */
2335
2362
  private async goDown(reason: string): Promise<ResidentDownError> {
2336
2363
  await this.setResidentState("down", reason);
2337
- this.deleteSchedules(REFRESH_CALLBACK);
2338
2364
  return new ResidentDownError(reason);
2339
2365
  }
2340
2366
 
@@ -2360,13 +2386,12 @@ export class ResidentDO extends Sandbox<Env> {
2360
2386
  await this.ctx.storage.delete([FACTS_KEY, SNAPSHOT_KEY]); // defensive: no stale facts from a past life
2361
2387
  this.deleteSchedules(PROVISIONING_CALLBACK);
2362
2388
  this.deleteSchedules(PROVISION_RUN_CALLBACK);
2363
- this.deleteSchedules(REFRESH_CALLBACK);
2364
2389
  await this.schedule(Math.max(1, Math.ceil(provisioningTimeoutMs / 1000)), PROVISIONING_CALLBACK, resource);
2365
2390
  await this.schedule(1, PROVISION_RUN_CALLBACK, resource);
2366
2391
  return { state: "onboarding", reason: "" };
2367
2392
  }
2368
2393
 
2369
- /** The provisioning engine (alarm-driven): clone bare mirror → resolve the
2394
+ /** The provisioning engine (schedule-driven): clone bare mirror → resolve the
2370
2395
  * default branch → full install + build in a working checkout using the
2371
2396
  * onboard-time command table → stamped snapshot → record facts → warm.
2372
2397
  * On failure: down(provision-failed at <step>) — the registry slot is
@@ -2481,7 +2506,8 @@ export class ResidentDO extends Sandbox<Env> {
2481
2506
  await this.writeDiskMarkers({ ready: sha, depsKey: lockfileHash, builtSha: sha });
2482
2507
  this.deleteSchedules(PROVISIONING_CALLBACK);
2483
2508
  await this.setResidentState("warm");
2484
- await this.armRefresh(resource);
2509
+ // The first refresh instance is the cron's: the row now reads `warm`
2510
+ // with no instance recorded, so the next firing creates it (item 9).
2485
2511
  } catch (err) {
2486
2512
  if ((await this.ctx.storage.get<ResidentState>(STATE_KEY)) !== "onboarding") return;
2487
2513
  this.deleteSchedules(PROVISIONING_CALLBACK);
@@ -2507,7 +2533,6 @@ export class ResidentDO extends Sandbox<Env> {
2507
2533
  await this.setResidentState("down", reason);
2508
2534
  this.deleteSchedules(PROVISIONING_CALLBACK);
2509
2535
  this.deleteSchedules(PROVISION_RUN_CALLBACK);
2510
- this.deleteSchedules(REFRESH_CALLBACK);
2511
2536
  const resource = await this.ctx.storage.get<string>(RESOURCE_KEY);
2512
2537
  if (resource) {
2513
2538
  try {
@@ -2525,12 +2550,13 @@ export class ResidentDO extends Sandbox<Env> {
2525
2550
  * otherwise still say warm while the R2 restore runs. Refuses mismatched
2526
2551
  * stamps → down(snapshot-stamp-mismatch); a stalled or capped restore →
2527
2552
  * down(r2-restore-failed); a restore the runtime replacement interrupts →
2528
- * degraded(restore-interrupted), rethrown so the cycle re-arms short.
2529
- * Throws ResidentDownError after the down transitions. Called by the
2530
- * refresh alarm (and the attach path). */
2553
+ * degraded(restore-interrupted), rethrown so the instance step that called
2554
+ * it is retried by the engine. Throws ResidentDownError after the down
2555
+ * transitions. Called by the refresh instance's fetch step (and the attach
2556
+ * path). */
2531
2557
  async ensureHydrated(): Promise<void> {
2532
2558
  // Fresh positive verdict for this incarnation → nothing to probe. See the
2533
- // per-incarnation memo block for why this is safe; the refresh alarm's
2559
+ // per-incarnation memo block for why this is safe; the refresh cycle's
2534
2560
  // 10-min cadence always outlives the TTL, so a cycle re-probes for real.
2535
2561
  if (this.hydratedVerdictAt !== 0 && systemClock() - this.hydratedVerdictAt < this.hydrationMemoTtlMs) return;
2536
2562
  if (this.hydration) return this.hydration;
@@ -2671,164 +2697,27 @@ export class ResidentDO extends Sandbox<Env> {
2671
2697
  await this.setResidentState("warm");
2672
2698
  }
2673
2699
 
2674
- // -- freshness (the self-rescheduling refresh alarm) --------------------------
2675
-
2676
- /** Self-rescheduling refresh: rehydrate if the container slept → mint a
2677
- * repo-scoped token (mint failure is command-level: recorded, never a
2678
- * lifecycle flip) → fetch into the bare mirror → when the default branch
2679
- * moved: plan against the disk checkpoints (planRefresh) — reuse a
2680
- * checkout an interrupted cycle already materialized, else update the
2681
- * checkout and reinstall ONLY if the committed lockfile key changed, then
2682
- * rebuild — write a new stamped snapshot, delete the replaced backup
2683
- * objects. Transitions: refreshing → warm, or degraded(reason) with the
2684
- * last snapshot still serving. */
2685
- async onRefreshAlarm(payload: string): Promise<void> {
2686
- // The freshness cycle nobody asked for is a root of its own
2687
- // (docs/reference/specs/tracing.md item 25): `resident.refresh`, with every command it
2688
- // ran as a `resident.<step>` child, exactly like an attach's.
2689
- const t0 = systemClock();
2690
- const trace = createStepTrace(t0);
2691
- let outcome = "ok";
2692
- try {
2693
- await this.stepTrace.run(trace, () => this.onRefreshAlarmTraced(payload));
2694
- } catch (err) {
2695
- outcome = "error";
2696
- throw err;
2697
- } finally {
2698
- emitStepRoot("resident.refresh", t0, trace.steps(), undefined, outcome);
2699
- }
2700
- }
2701
-
2702
- private async onRefreshAlarmTraced(payload: string): Promise<void> {
2703
- const resource = payload || ((await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "");
2704
- // Item 7: a resident on the Workflow lifecycle has no chain. An alarm a
2705
- // previous flip left armed — or an attach's +1 s pull, or a provisioning's
2706
- // first arm — runs nothing and re-arms nothing, so the two schedulers never
2707
- // both drive a cycle for one resident.
2708
- if ((await this.getLifecycle()) === "workflow") {
2709
- console.log(`refresh: lifecycle is workflow — the alarm chain runs no cycle for ${resource}`);
2710
- return;
2711
- }
2712
- let refreshCounted = false;
2713
- /** This cycle's lease holder in the in-flight row, once it is counted. */
2714
- let cycleHolder: string | null = null;
2715
- /** This firing's identity: what `fetchMirror` records so a repeated call inside the cycle is done. */
2716
- const cycle = crypto.randomUUID();
2717
- const before = await this.getStatus();
2718
- // down chains stay down (a rebuild is the escape hatch); onboarding is
2719
- // owned by provisioning, which arms the first refresh itself.
2720
- if (before.state === "onboarding" || before.state === "down") return;
2721
- try {
2722
- const gate = await this.refreshGate(resource);
2723
- if (!gate.go) return; // finally re-arms on the outcome the gate set
2724
- const { record, facts } = gate;
2725
- // From here the cycle mutates the mirror/checkout: count it as in flight
2726
- // so an attach-path reconcileImage never stops the container under it,
2727
- // and lease it in the in-flight row so the watchdog can tell this cycle
2728
- // from a marker a dead one left behind (item 22).
2729
- this.refreshesInFlight++;
2730
- refreshCounted = true;
2731
- cycleHolder = this.nextHolder();
2732
- await this.recordInFlight("refresh", cycleHolder, REFRESH_CYCLE_LEASE_MS, "refresh");
2733
-
2734
- const fetched = await this.refreshFetch(resource, facts, cycle, 1);
2735
- if (!fetched.ok) return;
2736
- const { sha, lockfileHash, mintError, token } = fetched;
2737
- const plan = planRefresh({
2738
- sha,
2739
- factsSha: facts.sha,
2740
- lockfileKey: lockfileHash,
2741
- disk: await this.readRefreshDisk(),
2742
- });
2743
- // Whether this cycle's snapshot step committed (`superseded` means
2744
- // another writer moved the record, whose facts then stand).
2745
- let committed = true;
2746
- if (plan.action !== "unchanged") {
2747
- const t0 = systemClock();
2748
- console.log(`refresh: ${facts.sha.slice(0, 8)} → ${sha.slice(0, 8)}: ${plan.action} (${plan.why})`);
2749
- // Deps come from the store (item 59): a changed lockfile key is
2750
- // materialized ONCE into `/workspace/deps/<key>` — OUTSIDE the mirror
2751
- // lock, because the install runs in its own scratch clone and touches
2752
- // no consumer's tree (the staging step) — and the checkout's
2753
- // node_modules becomes a hardlink view of that entry (runBuild). An
2754
- // attach that needs the same key joins this very install instead of
2755
- // starting its own. Checkpoint: the deps marker comes off BEFORE the
2756
- // install so an interruption mid-install can never read as completion.
2757
- let depsEntry: string | null = null;
2758
- if (plan.action === "rebuild") {
2759
- await this.refreshClearMarkers(plan.install);
2760
- if (plan.install) depsEntry = await this.refreshInstall(record, facts, sha, lockfileHash);
2761
- }
2762
- // `reuse`: the checkout already holds this sha with its deps and build
2763
- // (an interrupted cycle got that far) — the build step finds it done and
2764
- // only the snapshot, facts and stamp are missing; they move together.
2765
- await this.runBuild({
2766
- sha,
2767
- factsSha: facts.sha,
2768
- lockfileKey: lockfileHash,
2769
- buildCmd: record.commands.build,
2770
- depsEntry,
2771
- });
2772
- committed = await this.refreshSnapshot(resource, { ref: facts.defaultRef, sha, lockfileHash });
2773
- console.log(`refresh: ${sha.slice(0, 8)} ${plan.action} done in ${systemClock() - t0}ms`);
2774
- }
2775
- await this.refreshComplete(resource, facts, { sha, lockfileHash, committed, mintError, token });
2776
- } catch (err) {
2777
- if (err instanceof ResidentDownError) return; // already down with reason; chain stops below
2778
- const failure = await this.classifyCycleError(err);
2779
- // Set BEFORE the writes on purpose: if either throws, the finally still
2780
- // re-arms short — the safe direction for an interruption.
2781
- if (failure.interrupted) this.rearmOutcome = "interrupted";
2782
- await this.refreshFailed(failure, refreshCounted ? 1 : 0);
2783
- } finally {
2784
- if (refreshCounted) this.refreshesInFlight--;
2785
- if (cycleHolder) await this.clearInFlight("refresh", cycleHolder);
2786
- const state = await this.ctx.storage.get<ResidentState>(STATE_KEY);
2787
- // Consecutive-interruption count: bounds the short re-arm so
2788
- // a step whose output chronically carries the kill signature falls back to
2789
- // the cadence after INTERRUPTED_REARM_MAX_CONSECUTIVE instead of hot-looping.
2790
- let consecutiveInterrupted: number | undefined;
2791
- if (this.rearmOutcome === "interrupted") {
2792
- consecutiveInterrupted = ((await this.ctx.storage.get<number>(INTERRUPTED_STREAK_KEY)) ?? 0) + 1;
2793
- await this.ctx.storage.put(INTERRUPTED_STREAK_KEY, consecutiveInterrupted);
2794
- } else {
2795
- await this.ctx.storage.delete(INTERRUPTED_STREAK_KEY);
2796
- }
2797
- const interval = nextRefreshDelayS({
2798
- outcome: this.rearmOutcome,
2799
- intervalS: REFRESH_INTERVAL_S,
2800
- idleIntervalS: IDLE_REFRESH_INTERVAL_S,
2801
- consecutiveInterrupted,
2802
- });
2803
- this.rearmOutcome = "normal";
2804
- // A flip to `workflow` while this cycle ran: the chain ends here (item 7).
2805
- const chained = (await this.getLifecycle()) === "alarm";
2806
- if (chained && state && state !== "down" && state !== "onboarding") await this.armRefresh(resource, interval);
2807
- }
2808
- }
2809
-
2810
- // -- the cycle's phases, shared by the alarm and the instance (item 7) -------
2700
+ // -- freshness (the refresh cycle's phases, one per instance step) -----------
2811
2701
  //
2812
- // The alarm chain and the Workflow instance run the same cycle in the same
2813
- // order: the gates, the fetch, the plan, the install, the build, the
2814
- // snapshot, the completion. Each phase is one method here so neither
2815
- // scheduler carries a copy; the instance calls them one step at a time
2816
- // (`refreshInstance*`, below), the alarm in one handler (above).
2817
-
2818
- /** The cycle's entry gates, in the alarm's order: hydrate; the registry
2819
- * record (gone → the resident was offboarded mid-flight, nothing to do);
2820
- * the image reconcile (a stale image stops the container, which restarts
2821
- * on the current one — `image-stale-restart`); the disk-full re-probe (a
2822
- * disk still full decides its own recovery and stops the cycle — item 54);
2823
- * the park streak and the idle gate (`idle`); and, for a cycle that runs,
2824
- * the end of idle mode. Sets `rearmOutcome` for the alarm's finally; the
2825
- * instance resets it. */
2702
+ // The cycle runs as the Workflow instance in refresh.ts: the gates, the
2703
+ // fetch, the plan, the install, the build, the snapshot, the completion,
2704
+ // then the housekeeping steps. Each phase is one method here; the instance
2705
+ // calls them one step at a time (`refreshInstance*`, below).
2706
+
2707
+ /** The cycle's entry gates, in order: hydrate; the registry record (gone →
2708
+ * the resident was offboarded mid-flight, nothing to do); the image
2709
+ * reconcile (a stale image stops the container, which restarts on the
2710
+ * current one — thrown as `CycleRestartError` so the engine retries the
2711
+ * step onto it); the disk-full re-probe (a disk still full decides its own
2712
+ * recovery — a recycle is thrown the same way, a kept container stops the
2713
+ * cycle — item 54); the park streak and the idle gate (`idle`); and, for a
2714
+ * cycle that runs, the end of idle mode. */
2826
2715
  private async refreshGate(
2827
2716
  resource: string,
2828
2717
  ): Promise<{ go: false; why: string } | { go: true; record: ResidentRecord; facts: RepoFacts }> {
2829
2718
  await this.ensureHydrated();
2830
2719
  const record = await this.registry().getRecord(resource);
2831
- if (!record) return { go: false, why: "offboarded" }; // offboarded mid-flight: let the chain die quietly
2720
+ if (!record) return { go: false, why: "offboarded" }; // offboarded mid-flight: the cycle ends quietly
2832
2721
  const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
2833
2722
  if (!facts) throw new StepError("facts", "no repo facts recorded despite hydration");
2834
2723
 
@@ -2837,16 +2726,16 @@ export class ResidentDO extends Sandbox<Env> {
2837
2726
  // users the image lacks. Reconcile here (every cycle, cheap) — see
2838
2727
  // reconcileImage — so a rollout self-applies within one refresh.
2839
2728
  if (await this.reconcileImage("refresh")) {
2840
- // Container stopping; it restarts on the new image in seconds. Re-arm
2841
- // SHORT so the resident is re-warmed within a minute instead of
2842
- // sitting on the old cadence for a full 600 s.
2843
- this.rearmOutcome = "image-stale-restart";
2844
- return { go: false, why: "image-stale-restart" };
2729
+ // Container stopping; it restarts on the new image in seconds. The
2730
+ // engine's retry re-enters this step thirty seconds on and re-warms the
2731
+ // resident within the minute, instead of the next bucket.
2732
+ throw new CycleRestartError("image-stale-restart");
2845
2733
  }
2846
2734
 
2847
2735
  // Idle sleep: nobody has attached for IDLE_AFTER_S and no live tree is
2848
- // dirty → skip this fetch and park the alarm far out so SLEEP_AFTER can
2849
- // elapse. Staleness is repaid at the next attach (refreshIfStale). A
2736
+ // dirty → skip this fetch; the cron holds the next instance to the idle
2737
+ // cadence, so SLEEP_AFTER can elapse. Staleness is repaid at the next
2738
+ // attach (refreshIfStale). A
2850
2739
  // dirty live tree pins the container awake: sleep destroys the disk and
2851
2740
  // uncommitted work is not snapshotted.
2852
2741
  // Only a SETTLED resident may park: a cycle that finds `refreshing`/
@@ -2868,8 +2757,11 @@ export class ResidentDO extends Sandbox<Env> {
2868
2757
  // sweep freed trees) → run the cycle as usual and earn `warm`.
2869
2758
  const free = await this.freeKiB();
2870
2759
  if (free !== null && free < DISK_FULL_FREE_KIB) {
2871
- await this.recoverFromDiskFull(entry.reason, 0);
2872
- return { go: false, why: "disk-full" }; // the alarm's finally re-arms: short after a recycle, the cadence otherwise
2760
+ // A recycle: the engine's retry re-enters this step, restores onto the
2761
+ // empty disk and runs the cycle. A kept container: nothing a fetch can
2762
+ // do until space is freed; the cycle ends and the next bucket re-probes.
2763
+ if (await this.recoverFromDiskFull(entry.reason, 0)) throw new CycleRestartError("disk-full-restart");
2764
+ return { go: false, why: "disk-full" };
2873
2765
  }
2874
2766
  }
2875
2767
  let settled = entry.state === "warm";
@@ -2886,11 +2778,10 @@ export class ResidentDO extends Sandbox<Env> {
2886
2778
  await this.ctx.storage.put(DEGRADED_STREAK_KEY, streak);
2887
2779
  settled = streak.count >= DEGRADED_PARK_AFTER_CYCLES;
2888
2780
  } else {
2889
- // Warm, or a degraded stamped by the WATCHDOG (alarm-missed /
2890
- // stale-mid-flight) or by an INTERRUPTED cycle (refresh-interrupted —
2891
- // a deploy killed the step; it says nothing about the repo): the
2892
- // watchdog pulled this cycle to +5s precisely so a refresh RUNS, and the
2893
- // interrupted cycle re-armed short for the same reason.
2781
+ // Warm, or a degraded stamped by the WATCHDOG (stale-mid-flight) or by
2782
+ // an interrupted restore (restore-interrupted — a deploy rolled the
2783
+ // container; it says nothing about the repo): both mean a cycle must
2784
+ // RUN — this one.
2894
2785
  // Counting those toward the streak would be self-fulfilling —
2895
2786
  // each cycle that found the reason would park without attempting anything,
2896
2787
  // and after three the resident would sit parked-degraded for 6h at a
@@ -2905,8 +2796,7 @@ export class ResidentDO extends Sandbox<Env> {
2905
2796
  ...now,
2906
2797
  idleSince: new Date(systemClock()).toISOString(),
2907
2798
  } satisfies RepoFacts);
2908
- this.rearmOutcome = "idle";
2909
- return { go: false, why: "idle" }; // the alarm's finally re-arms at IDLE_REFRESH_INTERVAL_S
2799
+ return { go: false, why: "idle" }; // the cron creates the next instance at IDLE_REFRESH_INTERVAL_S
2910
2800
  }
2911
2801
  if (facts.idleSince) {
2912
2802
  const now = (await this.ctx.storage.get<RepoFacts>(FACTS_KEY)) ?? facts;
@@ -2924,9 +2814,9 @@ export class ResidentDO extends Sandbox<Env> {
2924
2814
  * the installation stays fresh, and a private one fails at the fetch
2925
2815
  * below into a visible `degraded(github-unreachable: …)`. Returning early
2926
2816
  * instead would freeze whatever state the resident was in — a public
2927
- * repo the App is not installed on would sit in the watchdog's
2928
- * `degraded(alarm-missed)` forever with an ever-staler mirror, because
2929
- * the App cannot mint for a repo it is not installed on. A failed fetch
2817
+ * repo the App is not installed on would sit `degraded` forever with an
2818
+ * ever-staler mirror, because the App cannot mint for a repo it is not
2819
+ * installed on. A failed fetch
2930
2820
  * is recorded here — `degraded` with its reason, the disk-full recovery
2931
2821
  * when that is the cause — and answered `ok: false`. */
2932
2822
  private async refreshFetch(
@@ -2957,7 +2847,7 @@ export class ResidentDO extends Sandbox<Env> {
2957
2847
  let sha: string;
2958
2848
  try {
2959
2849
  // Same mirror mutex as attach's fetch/worktree work: the
2960
- // refresh alarm and an in-flight attach serialize instead of racing
2850
+ // refresh cycle and an in-flight attach serialize instead of racing
2961
2851
  // a prune against a worktree clone.
2962
2852
  sha = (await this.fetchMirror({ ref: facts.defaultRef, cycle, token })).sha;
2963
2853
  } catch (err) {
@@ -3042,9 +2932,9 @@ export class ResidentDO extends Sandbox<Env> {
3042
2932
  * them in the same write as the record — a wake never sees a half-updated
3043
2933
  * pair; this is the cycle's own bookkeeping on a fresh read, and a
3044
2934
  * superseded snapshot leaves the other writer's stamp alone), `warm`, then
3045
- * the housekeeping that is never a lifecycle flip: the finished-ref
3046
- * reclamation the prune already informed (item 45) and the disk sample
3047
- * (item 55). */
2935
+ * the finished-ref reclamation the prune already informed (item 45) —
2936
+ * housekeeping, never a lifecycle flip. The disk sample is the instance's
2937
+ * own `measure` step (item 55), after its sweep. */
3048
2938
  private async refreshComplete(
3049
2939
  resource: string,
3050
2940
  facts: RepoFacts,
@@ -3082,27 +2972,21 @@ export class ResidentDO extends Sandbox<Env> {
3082
2972
  } catch (err) {
3083
2973
  console.log(`reclaim ${resource}: pass failed: ${errMsg(err)}`);
3084
2974
  }
3085
- // Item 55: the cycle's disk sample — what /residents, `repo list`, the
3086
- // watchdog line and the next attach admission read. Housekeeping too.
3087
- await this.measureDisk().catch((err) => console.log(`disk: measure failed: ${errMsg(err)}`));
3088
2975
  }
3089
2976
 
3090
2977
  /** What a cycle's throw means. A step killed from OUTSIDE (the container
3091
2978
  * replaced under it — an image-changing deploy or a container stop; a
3092
2979
  * Worker-only deploy leaves the container running and interrupts nothing)
3093
- * is `refresh-interrupted`: not evidence about the repo — it never counts
3094
- * toward the park streak (the entry gate) — and the chain re-arms SHORT so
3095
- * the resident is warm again within a minute instead of after the full
3096
- * cadence (an unclassified kill otherwise costs the resident the whole
3097
- * 10-minute cadence, e.g. `degraded(build-failed: exit 143 …)` until the
3098
- * next alarm); the instance throws it to the engine, whose retry re-enters
3099
- * the step. Any other failure is the repo's own: `<step>-failed: …` /
3100
- * `refresh-failed: …` as before. Non-StepErrors classify too — an SDK
3101
- * replacement error can surface between steps — with the generic
3102
- * "refresh" step, whose failure reason is the pre-existing
3103
- * `refresh-failed: …` shape. A full disk is a third class: `disk-full: …`,
3104
- * never serviceable, and the one failure the resident can act on itself
3105
- * (recoverFromDiskFull). */
2980
+ * is `refresh-interrupted`: not evidence about the repo, so it is never
2981
+ * recorded as `degraded` — the instance step throws it to the engine, whose
2982
+ * retry re-enters the same idempotent method (an unclassified kill would
2983
+ * instead cost the resident a `degraded(build-failed: exit 143 …)` until
2984
+ * the next cycle). Any other failure is the repo's own: `<step>-failed: …`
2985
+ * / `refresh-failed: …`. Non-StepErrors classify too — an SDK replacement
2986
+ * error can surface between steps — with the generic "refresh" step, whose
2987
+ * failure reason is the `refresh-failed: …` shape. A full disk is a third
2988
+ * class: `disk-full: …`, never serviceable, and the one failure the
2989
+ * resident can act on itself (recoverFromDiskFull). */
3106
2990
  private async classifyCycleError(err: unknown): Promise<RefreshFailure> {
3107
2991
  return err instanceof StepError
3108
2992
  ? await this.classifyFailure(err.step, err.message)
@@ -3127,56 +3011,38 @@ export class ResidentDO extends Sandbox<Env> {
3127
3011
 
3128
3012
  // -- the refresh cycle as a Workflow instance (item 7) --------------------------
3129
3013
  //
3130
- // `ResidentRefresh` (the Workflow entrypoint, refresh.ts) calls these four
3131
- // methods, one per step, through the DO stub. Each runs the same phase the
3132
- // alarm runs, over the same rows, so a step the engine retries re-enters
3014
+ // `ResidentRefresh` (the Workflow entrypoint, refresh.ts) calls these
3015
+ // methods, one per step, through the DO stub. Each runs one phase of the
3016
+ // cycle over the resident's rows, so a step the engine retries re-enters
3133
3017
  // the same idempotent read-then-act method (item 22) and finds the work
3134
3018
  // done. Inputs and answers are small facts — refs, shas, keys, a path, a
3135
3019
  // word — never a payload and never a credential: the token is minted inside
3136
3020
  // the step that needs it.
3137
3021
 
3138
- /** Which scheduler drives this resident's refresh cycle (LIFECYCLE_KEY). */
3022
+ /** The lifecycle row (LIFECYCLE_KEY): `workflow`, whatever a flagged rollout stored. */
3139
3023
  async getLifecycle(): Promise<ResidentLifecycle> {
3140
3024
  return lifecycleOf(await this.ctx.storage.get(LIFECYCLE_KEY));
3141
3025
  }
3142
3026
 
3143
- /** Flip the resident between the alarm chain and the Workflow instance
3144
- * (admin `/debug` `lifecycle`). To `workflow`: the pending alarm is
3145
- * dropped; an alarm already firing runs to its end and re-arms nothing.
3146
- * Back to `alarm`: the chain is re-armed the way the watchdog re-arms a
3147
- * dead one, when the resident is in a state the chain serves. */
3148
- async setLifecycle(mode: ResidentLifecycle): Promise<{ lifecycle: ResidentLifecycle; refreshSchedules: number }> {
3149
- const previous = await this.getLifecycle();
3027
+ /** Rewrite the lifecycle row (admin `/debug` `lifecycle`). `workflow` is
3028
+ * the one mode, so the op's only effect is to replace a stale `alarm`
3029
+ * value the flagged rollout left behind — the row then says what
3030
+ * `lifecycleOf` already reads. */
3031
+ async setLifecycle(mode: ResidentLifecycle): Promise<{ lifecycle: ResidentLifecycle }> {
3150
3032
  await this.ctx.storage.put(LIFECYCLE_KEY, mode);
3151
- const resource = (await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "";
3152
- if (mode === "workflow") {
3153
- this.deleteSchedules(REFRESH_CALLBACK);
3154
- } else if (previous !== "alarm") {
3155
- const { state } = await this.getStatus();
3156
- const pending = await this.listSchedules(REFRESH_CALLBACK);
3157
- if (state !== "onboarding" && state !== "down" && pending.length === 0) {
3158
- await this.schedule(5, REFRESH_CALLBACK, resource);
3159
- }
3160
- }
3161
- console.log(`lifecycle: ${resource} ${previous} → ${mode}`);
3162
- return { lifecycle: mode, refreshSchedules: (await this.listSchedules(REFRESH_CALLBACK)).length };
3033
+ return { lifecycle: mode };
3163
3034
  }
3164
3035
 
3165
3036
  private async instanceRow(): Promise<RefreshInstanceRow> {
3166
3037
  return (await this.ctx.storage.get<RefreshInstanceRow>(REFRESH_INSTANCE_KEY)) ?? { instance: null, skipped: null };
3167
3038
  }
3168
3039
 
3169
- /** The facts the cron's instance-creation decision reads
3170
- * (`shouldCreateRefreshInstance`): the flag, the state and when it last
3171
- * changed, idle mode, the last instance's creation time. */
3040
+ /** The facts the cron's instance-creation decision and the admin
3041
+ * `refresh-now` op read (`shouldCreateRefreshInstance`,
3042
+ * `refreshCycleBlocked`): the state and when it last changed, idle mode,
3043
+ * the last instance's creation time and whether the engine still runs it. */
3172
3044
  async refreshRow(): Promise<RefreshRow> {
3173
- const map = await this.ctx.storage.get<unknown>([
3174
- LIFECYCLE_KEY,
3175
- STATE_KEY,
3176
- UPDATED_KEY,
3177
- FACTS_KEY,
3178
- REFRESH_INSTANCE_KEY,
3179
- ]);
3045
+ const map = await this.ctx.storage.get<unknown>([STATE_KEY, UPDATED_KEY, FACTS_KEY, REFRESH_INSTANCE_KEY]);
3180
3046
  const facts = map.get(FACTS_KEY) as RepoFacts | undefined;
3181
3047
  const row = (map.get(REFRESH_INSTANCE_KEY) as RefreshInstanceRow | undefined) ?? { instance: null, skipped: null };
3182
3048
  const epochMs = (iso: string | undefined): number | null => {
@@ -3184,7 +3050,6 @@ export class ResidentDO extends Sandbox<Env> {
3184
3050
  return Number.isFinite(t) ? t : null;
3185
3051
  };
3186
3052
  return {
3187
- lifecycle: lifecycleOf(map.get(LIFECYCLE_KEY)),
3188
3053
  state: (map.get(STATE_KEY) as ResidentState | undefined) ?? "down",
3189
3054
  updatedAt: epochMs(map.get(UPDATED_KEY) as string | undefined),
3190
3055
  idleSince: epochMs(facts?.idleSince),
@@ -3193,27 +3058,28 @@ export class ResidentDO extends Sandbox<Env> {
3193
3058
  };
3194
3059
  }
3195
3060
 
3196
- /** Whether the engine still runs `id`: queued, running, paused or waiting.
3197
- * The one fact the marker's age cannot give — a step between retry
3198
- * attempts holds no lease and writes nothing — read from the engine, which
3199
- * knows. An unknown id, a missing binding or a failed read answer false:
3200
- * the marker's age then decides, as before. */
3201
- private async instanceRunning(id: string | null): Promise<boolean> {
3202
- if (!id) return false;
3061
+ /** The engine's own word on the instance `id` — `queued`, `running`,
3062
+ * `complete`, `errored`, … — or null for an id the engine does not know (a
3063
+ * missing binding or a failed read answer the same). The one fact the
3064
+ * marker's age cannot give — a step between retry attempts holds no lease
3065
+ * and writes nothing — read from the engine, which knows. */
3066
+ private async instanceStatus(id: string | null): Promise<string | null> {
3067
+ if (!id) return null;
3203
3068
  try {
3204
- const { status } = await (await this.env.RESIDENT_REFRESH.get(id)).status();
3205
- return (
3206
- status === "queued" ||
3207
- status === "running" ||
3208
- status === "paused" ||
3209
- status === "waiting" ||
3210
- status === "waitingForPause"
3211
- );
3069
+ return (await (await this.env.RESIDENT_REFRESH.get(id)).status()).status;
3212
3070
  } catch {
3213
- return false;
3071
+ return null;
3214
3072
  }
3215
3073
  }
3216
3074
 
3075
+ /** Whether the engine still runs `id`: queued, running, paused or waiting.
3076
+ * Anything else — a finished or failed instance, an unknown id — is not a
3077
+ * live cycle, and the marker's age then decides, as before. */
3078
+ private async instanceRunning(id: string | null): Promise<boolean> {
3079
+ const status = await this.instanceStatus(id);
3080
+ return status !== null && INSTANCE_LIVE_STATUSES.has(status);
3081
+ }
3082
+
3217
3083
  /** The cron created an instance for this resident. */
3218
3084
  async recordRefreshInstance(id: string, createdAtMs: number): Promise<void> {
3219
3085
  const row = await this.instanceRow();
@@ -3288,28 +3154,39 @@ export class ResidentDO extends Sandbox<Env> {
3288
3154
  } satisfies RefreshInstanceRow);
3289
3155
  }
3290
3156
 
3291
- /** Run one step of the refresh instance the way the alarm runs its cycle:
3292
- * counted in flight (so an attach-path reconcileImage never stops the
3293
- * container under it), under a step trace the instance grafts on its root,
3157
+ /** Run one step of the refresh instance: counted in flight once past the
3158
+ * entry gates (so an attach-path reconcileImage never stops the container
3159
+ * under it — while the gates themselves, `isIdle`, `reconcileImage("refresh")`
3160
+ * and the disk-full recycle, must not see the probing cycle as an operation
3161
+ * in flight, or no resident would ever park, restart a stale image or
3162
+ * recycle a full disk; the step is handed `cycle.count` and calls it once
3163
+ * its gates have passed — the fetch step after `refreshGate`, every other
3164
+ * step first thing), under a step trace the instance grafts on its root,
3294
3165
  * its outcome on the instance row for `/status`. A step killed from outside
3295
- * (the container replaced under it) is thrown to the engine, whose retry
3166
+ * (the container replaced under it), or a gate that stopped the container
3167
+ * on purpose (`CycleRestartError`), is thrown to the engine, whose retry
3296
3168
  * re-enters the same idempotent method — the row stays `refreshing`, never
3297
3169
  * `degraded`, and a `refreshing` younger than the stale bound keeps the
3298
3170
  * cron from creating a second instance meanwhile. A failure of the repo's
3299
- * own is recorded as the alarm records it — `degraded` with the reason,
3300
- * the last snapshot still serving — and answered `failed`, which ends the
3301
- * instance; the next cron firing starts the next cycle from that state. */
3171
+ * own is recorded — `degraded` with the reason, the last snapshot still
3172
+ * serving — and answered `failed`, which ends the cycle; the next cron
3173
+ * firing starts the next one from that state. */
3302
3174
  private async runInstanceStep<T>(
3303
3175
  instance: string,
3304
3176
  step: string,
3305
- fn: () => Promise<InstanceStepResult<T>>,
3177
+ fn: (cycle: { count: () => void }) => Promise<InstanceStepResult<T>>,
3306
3178
  ): Promise<InstanceStepAnswer<T>> {
3307
3179
  const startedAt = systemClock();
3308
3180
  const trace = createStepTrace(startedAt);
3309
- this.refreshesInFlight++;
3181
+ let counted = false;
3182
+ const count = () => {
3183
+ if (counted) return;
3184
+ counted = true;
3185
+ this.refreshesInFlight++;
3186
+ };
3310
3187
  let outcome = "done";
3311
3188
  try {
3312
- const result = await this.stepTrace.run(trace, fn);
3189
+ const result = await this.stepTrace.run(trace, () => fn({ count }));
3313
3190
  if (result.status !== "done") {
3314
3191
  outcome = result.status === "stopped" ? `stopped (${result.why})` : `failed (${result.reason})`;
3315
3192
  await this.clearInstanceLease(instance);
@@ -3322,20 +3199,27 @@ export class ResidentDO extends Sandbox<Env> {
3322
3199
  await this.clearInstanceLease(instance);
3323
3200
  return { status: "failed", reason: err.message, startedAt, trace: trace.steps() };
3324
3201
  }
3202
+ if (err instanceof CycleRestartError) {
3203
+ outcome = `restarting (${err.why}) — the engine retries`;
3204
+ console.log(`refresh instance ${instance}: ${step} ${err.message}`);
3205
+ throw err;
3206
+ }
3325
3207
  const failure = await this.classifyCycleError(err);
3326
3208
  if (failure.interrupted) {
3327
3209
  outcome = `interrupted (${failure.reason}) — the engine retries`;
3328
3210
  console.log(`refresh instance ${instance}: ${step} interrupted — ${failure.reason.slice(0, 400)}; retrying`);
3329
3211
  throw err;
3330
3212
  }
3331
- await this.refreshFailed(failure, 1);
3213
+ // The disk-full recovery excludes this cycle from its in-flight count only
3214
+ // when the cycle counted itself: a failure inside the gates (before
3215
+ // `cycle.count()`) contributed nothing, and excluding it anyway would read
3216
+ // one real run as none and recycle the container under it.
3217
+ await this.refreshFailed(failure, counted ? 1 : 0);
3332
3218
  await this.clearInstanceLease(instance);
3333
3219
  outcome = `failed (${failure.reason})`;
3334
3220
  return { status: "failed", reason: failure.reason, startedAt, trace: trace.steps() };
3335
3221
  } finally {
3336
- this.refreshesInFlight--;
3337
- // The gates set this for the alarm's finally; no alarm runs on this path.
3338
- this.rearmOutcome = "normal";
3222
+ if (counted) this.refreshesInFlight--;
3339
3223
  await this.recordInstanceStep(instance, step, outcome).catch((err) =>
3340
3224
  console.log(`refresh instance ${instance}: recording ${step} failed: ${errMsg(err)}`),
3341
3225
  );
@@ -3350,12 +3234,17 @@ export class ResidentDO extends Sandbox<Env> {
3350
3234
  resource: string;
3351
3235
  instance: string;
3352
3236
  }): Promise<InstanceStepAnswer<RefreshFetchFacts>> {
3353
- return this.runInstanceStep<RefreshFetchFacts>(input.instance, "fetch", async () => {
3237
+ return this.runInstanceStep<RefreshFetchFacts>(input.instance, "fetch", async (cycle) => {
3354
3238
  const before = await this.getStatus();
3355
3239
  // down stays down (a rebuild is the escape hatch); onboarding is owned by provisioning.
3356
3240
  if (before.state === "onboarding" || before.state === "down") return { status: "stopped", why: "not-serving" };
3357
3241
  const gate = await this.refreshGate(input.resource);
3358
3242
  if (!gate.go) return { status: "stopped", why: gate.why };
3243
+ // Past the gates the cycle mutates the mirror and checkout: count it in
3244
+ // flight from here — never before, or the gates above (the idle park,
3245
+ // the image reconcile, the disk-full recycle) would have seen this
3246
+ // cycle as a live operation and never fired.
3247
+ cycle.count();
3359
3248
  const { record, facts } = gate;
3360
3249
  const holder = this.nextHolder();
3361
3250
  await this.recordInFlight("refresh", holder, REFRESH_CYCLE_LEASE_MS, "refresh");
@@ -3394,7 +3283,8 @@ export class ResidentDO extends Sandbox<Env> {
3394
3283
  sha: string;
3395
3284
  lockfileKey: string;
3396
3285
  }): Promise<InstanceStepAnswer<{ entry: string | null }>> {
3397
- return this.runInstanceStep<{ entry: string | null }>(input.instance, "install", async () => {
3286
+ return this.runInstanceStep<{ entry: string | null }>(input.instance, "install", async (cycle) => {
3287
+ cycle.count();
3398
3288
  const record = await this.registry().getRecord(input.resource);
3399
3289
  if (!record) return { status: "stopped", why: "offboarded" };
3400
3290
  const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
@@ -3414,7 +3304,8 @@ export class ResidentDO extends Sandbox<Env> {
3414
3304
  lockfileKey: string;
3415
3305
  depsEntry: string | null;
3416
3306
  }): Promise<InstanceStepAnswer<{ ran: boolean; why: string }>> {
3417
- return this.runInstanceStep<{ ran: boolean; why: string }>(input.instance, "build", async () => {
3307
+ return this.runInstanceStep<{ ran: boolean; why: string }>(input.instance, "build", async (cycle) => {
3308
+ cycle.count();
3418
3309
  const record = await this.registry().getRecord(input.resource);
3419
3310
  if (!record) return { status: "stopped", why: "offboarded" };
3420
3311
  const built = await this.runBuild({
@@ -3432,9 +3323,9 @@ export class ResidentDO extends Sandbox<Env> {
3432
3323
 
3433
3324
  /** Step `snapshot`: the stamped pair to R2 (`snapshot` finds a record at the
3434
3325
  * stamp done and answers `superseded` to another writer, never a throw),
3435
- * then the cycle's completion — facts, `warm`, the reclamation, the disk
3436
- * sample — and the cycle lease released. An `unchanged` cycle skips the
3437
- * archive and still completes, as the alarm does. */
3326
+ * then the cycle's completion — facts, `warm`, the reclamation — and the
3327
+ * cycle lease released. An `unchanged` cycle skips the archive and still
3328
+ * completes. */
3438
3329
  async refreshInstanceSnapshot(input: {
3439
3330
  resource: string;
3440
3331
  instance: string;
@@ -3444,7 +3335,8 @@ export class ResidentDO extends Sandbox<Env> {
3444
3335
  action: RefreshPlan["action"];
3445
3336
  mintError: string | null;
3446
3337
  }): Promise<InstanceStepAnswer<{ committed: boolean }>> {
3447
- return this.runInstanceStep<{ committed: boolean }>(input.instance, "snapshot", async () => {
3338
+ return this.runInstanceStep<{ committed: boolean }>(input.instance, "snapshot", async (cycle) => {
3339
+ cycle.count();
3448
3340
  const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
3449
3341
  if (!facts) return { status: "stopped", why: "no-facts" };
3450
3342
  const committed =
@@ -3480,11 +3372,50 @@ export class ResidentDO extends Sandbox<Env> {
3480
3372
  });
3481
3373
  }
3482
3374
 
3483
- /** Set during one alarm by the idle gate (`idle`), an image-stale container
3484
- * stop (`image-stale-restart`), a disk-full recycle (`disk-full-restart`)
3485
- * or an interrupted step (`interrupted`) so `finally` picks the matching
3486
- * re-arm delay (`nextRefreshDelayS`); reset to `normal` after every arm. */
3487
- private rearmOutcome: RefreshOutcome = "normal";
3375
+ /** A housekeeping step never flips lifecycle state: its failure is a log
3376
+ * line and a `done` answer naming it (`result` null) — except a step killed
3377
+ * from outside, which the engine retries like any other. */
3378
+ private async housekeeping<T>(
3379
+ step: string,
3380
+ work: () => Promise<T>,
3381
+ ): Promise<InstanceStepResult<{ result: T | null; error: string | null }>> {
3382
+ try {
3383
+ return { status: "done", result: await work(), error: null };
3384
+ } catch (err) {
3385
+ const failure = await this.classifyCycleError(err);
3386
+ if (failure.interrupted) throw err;
3387
+ console.log(`refresh instance: ${step} failed — ${failure.reason.slice(0, 400)}`);
3388
+ return { status: "done", result: null, error: residentText(failure.reason) };
3389
+ }
3390
+ }
3391
+
3392
+ /** Step `sweep`: the worktree inactivity sweep (item 23) — TTL eviction and
3393
+ * the clean-idle release. No gate runs here, so the step counts its cycle
3394
+ * first thing, like install, build and snapshot. Idempotent by shape: a
3395
+ * second call finds the bindings it evicted already evicted and nothing
3396
+ * else past its cutoffs. */
3397
+ async refreshInstanceSweep(input: {
3398
+ resource: string;
3399
+ instance: string;
3400
+ }): Promise<InstanceStepAnswer<{ result: { evicted: string[]; kept: number } | null; error: string | null }>> {
3401
+ return this.runInstanceStep(input.instance, "sweep", async (cycle) => {
3402
+ cycle.count();
3403
+ return this.housekeeping("sweep", () => this.sweepWorktrees(input.resource));
3404
+ });
3405
+ }
3406
+
3407
+ /** Step `measure`: the disk sample (item 55) — one `df` + one `du`, written
3408
+ * over the last; a second call takes the same sample again. Counts its
3409
+ * cycle first thing, like every step past the gates. Never wakes a slept
3410
+ * container (`measureDisk` answers null, `measured: false`). */
3411
+ async refreshInstanceMeasure(input: {
3412
+ instance: string;
3413
+ }): Promise<InstanceStepAnswer<{ result: { measured: boolean } | null; error: string | null }>> {
3414
+ return this.runInstanceStep(input.instance, "measure", async (cycle) => {
3415
+ cycle.count();
3416
+ return this.housekeeping("measure", async () => ({ measured: (await this.measureDisk()) !== null }));
3417
+ });
3418
+ }
3488
3419
 
3489
3420
  /** Idle = no live binding attached within IDLE_AFTER_S AND (when the
3490
3421
  * runtime is up) no live tree is dirty. Bindings are storage; dirtiness
@@ -3538,14 +3469,15 @@ export class ResidentDO extends Sandbox<Env> {
3538
3469
  return classifyRefreshFailure({ step, message, freeKiB: await this.freeKiB() });
3539
3470
  }
3540
3471
 
3541
- /** The disk is a cache: stop the container so the next alarm restores
3542
- * mirror + checkout from R2 onto an empty disk — the same wake path as a
3543
- * platform sleep. Only when the pure plan allows it: nothing in flight
3544
- * (`selfInFlight` excludes the calling refresh cycle from the count), every
3545
- * live tree clean, and no recycle within the cooldown. A refused recycle is
3546
- * written to `lastRefreshError` with its why, so `/residents` says what an
3547
- * operator must do; the `degraded` reason stays the clean `disk-full: …`. */
3548
- private async recoverFromDiskFull(reason: string, selfInFlight: number): Promise<void> {
3472
+ /** The disk is a cache: stop the container so the next cycle attempt
3473
+ * restores mirror + checkout from R2 onto an empty disk — the same wake
3474
+ * path as a platform sleep. Only when the pure plan allows it: nothing in
3475
+ * flight (`selfInFlight` excludes the calling refresh cycle from the
3476
+ * count), every live tree clean, and no recycle within the cooldown. A
3477
+ * refused recycle is written to `lastRefreshError` with its why, so
3478
+ * `/residents` says what an operator must do; the `degraded` reason stays
3479
+ * the clean `disk-full: …`. Answers whether the container was recycled. */
3480
+ private async recoverFromDiskFull(reason: string, selfInFlight: number): Promise<boolean> {
3549
3481
  const lastRecycleAt = await this.ctx.storage.get<number>(DISK_FULL_RECYCLE_KEY);
3550
3482
  const plan = planDiskFullRecovery({
3551
3483
  now: systemClock(),
@@ -3556,16 +3488,16 @@ export class ResidentDO extends Sandbox<Env> {
3556
3488
  if (plan.action === "wait") {
3557
3489
  console.log(`disk-full: container kept — ${plan.why}`);
3558
3490
  await this.recordRefreshError(`${reason} — container kept: ${plan.why}`);
3559
- return;
3491
+ return false;
3560
3492
  }
3561
3493
  console.log(
3562
- `disk-full: recycling the container — the next alarm restores mirror + checkout from R2 onto an empty disk (${reason})`,
3494
+ `disk-full: recycling the container — the next cycle restores mirror + checkout from R2 onto an empty disk (${reason})`,
3563
3495
  );
3564
3496
  await this.ctx.storage.put(DISK_FULL_RECYCLE_KEY, systemClock());
3565
- await this.recordRefreshError(`${reason} — container recycled; restoring from R2 on the next alarm`);
3497
+ await this.recordRefreshError(`${reason} — container recycled; restoring from R2 on the next cycle`);
3566
3498
  this.swapIncarnation(); // deliberate incarnation swap
3567
3499
  await this.stop().catch((err) => console.log(`disk-full: stop failed: ${errMsg(err)}`));
3568
- this.rearmOutcome = "disk-full-restart";
3500
+ return true;
3569
3501
  }
3570
3502
 
3571
3503
  // -- disk budget (docs/reference/specs/resident-repos.md item 55) -------------------------
@@ -3625,18 +3557,6 @@ export class ResidentDO extends Sandbox<Env> {
3625
3557
  return sample;
3626
3558
  }
3627
3559
 
3628
- /** Schedule callback: the deferred measurement an attach/detach/sweep arms. */
3629
- async onDiskMeasure(_payload: string): Promise<void> {
3630
- await this.measureDisk().catch((err) => console.log(`disk: measure failed: ${errMsg(err)}`));
3631
- }
3632
-
3633
- /** Arm one deferred measurement (at most one pending): the attach's hot path
3634
- * pays a `df`, not the `du`. */
3635
- private async scheduleDiskMeasure(resource: string): Promise<void> {
3636
- if ((await this.listSchedules(DISK_MEASURE_CALLBACK)).length > 0) return;
3637
- await this.schedule(DISK_MEASURE_DELAY_S, DISK_MEASURE_CALLBACK, resource);
3638
- }
3639
-
3640
3560
  /** `{usedKiB, totalKiB, at}` of the last sample for the watchdog line (storage
3641
3561
  * only — the watchdog never touches the container). */
3642
3562
  private async diskGauge(): Promise<{ usedKiB: number; totalKiB: number; freeKiB: number; at: string } | null> {
@@ -3781,9 +3701,10 @@ export class ResidentDO extends Sandbox<Env> {
3781
3701
 
3782
3702
  /** Refresh-on-attach, BOUNDED: if the resident was idle (or the last
3783
3703
  * refresh is older than the active cadence), fetch the mirror now — seconds,
3784
- * under the mirror lock — so the ref this thread binds is current, clear
3785
- * idle mode, and pull the full refresh cycle (checkout rebuild if main
3786
- * moved: minutes) to +1s in the BACKGROUND. The attach never waits on an
3704
+ * under the mirror lock — so the ref this thread binds is current, and
3705
+ * clear idle mode, which puts the full refresh cycle (checkout rebuild if
3706
+ * main moved: minutes) on the cron's awake cadence — the next firing
3707
+ * creates its instance, within one bucket. The attach never waits on an
3787
3708
  * install/build, so a wake cannot become a cold-fallback generator. */
3788
3709
  private async refreshIfStale(resource: string): Promise<void> {
3789
3710
  const facts = await this.ctx.storage.get<RepoFacts>(FACTS_KEY);
@@ -3798,7 +3719,7 @@ export class ResidentDO extends Sandbox<Env> {
3798
3719
  // Bounded like attach's own clone section: a full checkout rebuild
3799
3720
  // holding the mutex must not stall a wake attach for minutes — on
3800
3721
  // expiry (MirrorBusyError) proceed on the last mirror, same as a failed
3801
- // fetch. The background full cycle armed below repays the staleness.
3722
+ // fetch. The next refresh instance repays the staleness.
3802
3723
  await this.withMirrorLock(
3803
3724
  () =>
3804
3725
  this.gitWithCred(
@@ -3822,7 +3743,6 @@ export class ResidentDO extends Sandbox<Env> {
3822
3743
  const { idleSince: _woke, ...awake } = fresh;
3823
3744
  await this.ctx.storage.put(FACTS_KEY, awake satisfies RepoFacts);
3824
3745
  }
3825
- await this.armRefresh(resource, 1); // full cycle now, in the background; it re-arms at the active cadence
3826
3746
  }
3827
3747
 
3828
3748
  /** Pool users live in the IMAGE (Dockerfile useradd loop) while THREAD_USERS
@@ -3851,20 +3771,26 @@ export class ResidentDO extends Sandbox<Env> {
3851
3771
  return true;
3852
3772
  }
3853
3773
 
3854
- // -- watchdog (the sparse cron that re-arms dead alarm chains) ---------------
3855
-
3856
- /** One watchdog pass over this resident (invoked by the Worker cron):
3857
- * re-arm a dead refresh chain and mark degraded(alarm-missed); time out an
3858
- * onboarding stuck past its budget → down(provision-timeout) + cap slot
3859
- * release; auto-rebuild a resident stuck down on unusable snapshots (one
3860
- * strike per pass, rebuild at AUTO_REBUILD_AFTER_STRIKES). Storage/
3861
- * schedule reads (plus the strike counter) only — containers start via the
3862
- * re-armed alarms, never in this pass. */
3774
+ // -- watchdog (the sparse cron; it re-arms nothing) --------------------------
3775
+
3776
+ /** One watchdog pass over this resident (invoked by the Worker cron): time
3777
+ * out an onboarding stuck past its budget → down(provision-timeout) + cap
3778
+ * slot release; auto-rebuild a resident stuck down on unusable snapshots
3779
+ * (one strike per pass, rebuild at AUTO_REBUILD_AFTER_STRIKES); name a
3780
+ * `refreshing`/`restoring` marker older than STALE_MIDFLIGHT_MS that no
3781
+ * live lease and no running instance stands behind —
3782
+ * `degraded(stale-mid-flight: …)` carrying the last instance's id and the
3783
+ * engine's word on it, so a failed instance is visible by name on
3784
+ * `/status` (item 9) — and the next instance the cron creates (this very
3785
+ * pass) normalizes it. Storage and engine-status reads only: containers
3786
+ * are started by the instances' steps, never in this pass. The refresh row
3787
+ * the cron's instance-creation decision reads rides on the answer, read
3788
+ * after the check settled the state. */
3863
3789
  async watchdogCheck(): Promise<{
3864
3790
  resource: string;
3865
3791
  state: ResidentState;
3866
3792
  reason: string;
3867
- action: "none" | "rearmed" | "provision-timed-out" | "auto-rebuilt";
3793
+ action: "none" | "provision-timed-out" | "auto-rebuilt";
3868
3794
  /** Item 55: the last disk sample's gauge, for the watchdog's status line. */
3869
3795
  disk: { usedKiB: number; totalKiB: number; freeKiB: number; at: string } | null;
3870
3796
  /** Item 7: what the cron's instance-creation decision reads, after the check above settled the state. */
@@ -3878,7 +3804,7 @@ export class ResidentDO extends Sandbox<Env> {
3878
3804
  resource: string;
3879
3805
  state: ResidentState;
3880
3806
  reason: string;
3881
- action: "none" | "rearmed" | "provision-timed-out" | "auto-rebuilt";
3807
+ action: "none" | "provision-timed-out" | "auto-rebuilt";
3882
3808
  }> {
3883
3809
  const resource = (await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "";
3884
3810
  const status = await this.getStatus();
@@ -3893,8 +3819,8 @@ export class ResidentDO extends Sandbox<Env> {
3893
3819
  }
3894
3820
  if (status.state === "down") {
3895
3821
  // Auto-rebuild escape hatch: only rehydration-flavored downs — the
3896
- // snapshots themselves are the problem, and down chains never retry, so
3897
- // without this the resident would stay down forever.
3822
+ // snapshots themselves are the problem, and a down resident runs no
3823
+ // cycle, so without this the resident would stay down forever.
3898
3824
  if (REHYDRATION_FAILURE_RE.test(status.reason)) {
3899
3825
  const strikes = ((await this.ctx.storage.get<number>(REBUILD_STRIKES_KEY)) ?? 0) + 1;
3900
3826
  if (strikes >= AUTO_REBUILD_AFTER_STRIKES) {
@@ -3914,51 +3840,16 @@ export class ResidentDO extends Sandbox<Env> {
3914
3840
  // Any serving state clears accumulated strikes (a recovery must reset the
3915
3841
  // counter, or an unrelated later down inherits stale strikes).
3916
3842
  await this.ctx.storage.delete(REBUILD_STRIKES_KEY);
3917
- // Item 7: a resident on the Workflow lifecycle has no chain to re-arm —
3918
- // its cycles are the instances the cron creates. The sweep re-arm below
3919
- // is housekeeping either way; the two refresh re-arms are the alarm's.
3920
- const chained = (await this.getLifecycle()) === "alarm";
3921
-
3922
- // The sweep chain has the same failure mode as the refresh chain (a DO
3923
- // eviction mid-callback kills the self-rescheduling), but nothing re-armed
3924
- // it: only an attach did, so a resident with live bindings and no traffic
3925
- // never swept again (idle bindings sat for hours with no sweep).
3926
- // Re-arm at +5s whenever live bindings exist and none is pending. Not a
3927
- // lifecycle event — the sweep is housekeeping, no state flip. Runs BEFORE
3928
- // the stale-mid-flight check so that branch's early return never skips it.
3929
- const bindings = await this.ctx.storage.list<ThreadBinding>({ prefix: THREAD_KEY_PREFIX });
3930
- const liveBindings = [...bindings.values()].some((b) => !b.evicted && b.user);
3931
- // `sweepInFlight` is the explicit guard against arming a second chain while
3932
- // a sweep is executing (its schedule row also stays listed until the callback
3933
- // resolves, but that is a library detail we do not lean on).
3934
- if (liveBindings && !this.sweepInFlight) {
3935
- const pendingSweeps = await this.listSchedules(SWEEP_CALLBACK);
3936
- // Config drift: a row armed by OLDER code (e.g. a daily sweep from
3937
- // before the cadence shortened, due 24h out) is still honored by the runtime, so a
3938
- // shorter SWEEP_INTERVAL_S never takes effect until it fires. Treat a row
3939
- // due further out than the current interval (+ slack) as stale and
3940
- // replace it, so a deploy that shortens the cadence applies within one
3941
- // watchdog pass rather than after the old delay elapses.
3942
- const nowS = Math.floor(systemClock() / 1000);
3943
- const drifted = pendingSweeps.some((row) => (row.time ?? 0) - nowS > SWEEP_INTERVAL_S + SWEEP_DRIFT_SLACK_S);
3944
- // Re-check the guard: listSchedules yielded, and a sweep that started
3945
- // meanwhile owns the row its own `finally` is about to arm.
3946
- if ((pendingSweeps.length === 0 || drifted) && !this.sweepInFlight) {
3947
- this.deleteSchedules(SWEEP_CALLBACK);
3948
- await this.schedule(5, SWEEP_CALLBACK, resource);
3949
- // Disjoint by construction: inside this branch, a non-empty list implies `drifted`.
3950
- console.log(
3951
- `watchdog ${resource}: sweep ${pendingSweeps.length === 0 ? "chain was dead" : "row was due beyond the current interval (config drift)"} with live bindings — re-armed`,
3952
- );
3953
- }
3954
- }
3955
3843
 
3956
3844
  // A mid-flight state older than STALE_MIDFLIGHT_MS with no cycle or restore
3957
- // actually running is a marker orphaned by an interrupted cycle (DO evicted
3958
- // by a deploy, platform restart). Left alone it is permanent — the idle gate
3959
- // above only parks from `warm`, but nothing else would ever rewrite it, and
3960
- // the bot's warm-gate keeps sending runs cold. Mark it degraded (visible —
3961
- // named degradation, never a stall) and pull the next cycle to +5s so it normalizes.
3845
+ // actually running is a marker orphaned by a cycle that died — a DO evicted
3846
+ // by a deploy, a platform restart, an instance out of retries mid-step.
3847
+ // Left alone it is permanent — the idle gate only parks from `warm`, and
3848
+ // nothing else would ever rewrite it, so the bot's warm-gate keeps sending
3849
+ // runs cold. Mark it degraded (visible — named degradation, never a
3850
+ // stall), naming the instance the row last recorded and the engine's word
3851
+ // on it; the marker is no longer `refreshing`, so the instance the cron
3852
+ // creates in this same pass normalizes it.
3962
3853
  if (status.state === "refreshing" || status.state === "restoring") {
3963
3854
  const updatedAt = Date.parse((await this.ctx.storage.get<string>(UPDATED_KEY)) ?? "") || 0;
3964
3855
  // Who holds what comes from the in-flight row (item 22), not from this
@@ -3980,49 +3871,32 @@ export class ResidentDO extends Sandbox<Env> {
3980
3871
  if (again.state !== status.state || liveAgain.refresh || liveAgain.hydration) {
3981
3872
  return { resource, ...again, action: "none" };
3982
3873
  }
3983
- // Drop the dead hydration reference and the dead leases so the
3984
- // re-armed cycle's ensureHydrated starts a fresh restore instead of
3985
- // awaiting a promise that will never settle. Safe: past the bound
3986
- // nothing on the other end is still writing (the container it talked
3987
- // to is gone). Each clear is compared against the holder just read:
3988
- // a cycle that recorded a fresh lease between that read and this
3989
- // delete keeps it, the way a release never deletes another holder's row.
3990
- this.hydration = null;
3991
- if (!chained && (await this.instanceRunning((await this.instanceRow()).instance?.id ?? null))) {
3874
+ const last = (await this.instanceRow()).instance?.id ?? null;
3875
+ const engine = await this.instanceStatus(last);
3876
+ if (engine !== null && INSTANCE_LIVE_STATUSES.has(engine)) {
3992
3877
  // The engine still runs the recorded instance — a step between retry
3993
3878
  // attempts, holding no lease and writing no state. Not stale: leave
3994
3879
  // the marker, create nothing (the cron's decision reads the same fact).
3995
3880
  return { resource, ...status, action: "none" };
3996
3881
  }
3882
+ // Drop the dead hydration reference and the dead leases so the next
3883
+ // cycle's ensureHydrated starts a fresh restore instead of awaiting a
3884
+ // promise that will never settle. Safe: past the bound nothing on the
3885
+ // other end is still writing (the container it talked to is gone).
3886
+ // Each clear is compared against the holder just read: a cycle that
3887
+ // recorded a fresh lease between that read and this delete keeps it,
3888
+ // the way a release never deletes another holder's row.
3889
+ this.hydration = null;
3997
3890
  if (rowAgain.refresh) await this.clearInFlight("refresh", rowAgain.refresh.holder);
3998
3891
  if (rowAgain.hydration) await this.clearInFlight("hydration", rowAgain.hydration.holder);
3999
- if (!chained) {
4000
- // The orphan is named the same way; the next instance the cron
4001
- // creates (this very pass — the marker is no longer `refreshing`)
4002
- // normalizes it, no alarm involved.
4003
- const reason = `stale-mid-flight: ${status.state} since ${new Date(updatedAt).toISOString()} with no cycle running; the next refresh instance normalizes it`;
4004
- await this.setResidentState("degraded", reason);
4005
- return { resource, state: "degraded", reason, action: "none" };
4006
- }
4007
- const reason = `stale-mid-flight: ${status.state} since ${new Date(updatedAt).toISOString()} with no cycle running; re-armed by watchdog`;
3892
+ const instance = last
3893
+ ? `the last instance ${last} is ${engine ?? "unknown to the engine"}`
3894
+ : "no instance recorded";
3895
+ const reason = `stale-mid-flight: ${status.state} since ${new Date(updatedAt).toISOString()} with no cycle running; ${instance}; the next refresh instance normalizes it`;
4008
3896
  await this.setResidentState("degraded", reason);
4009
- this.deleteSchedules(REFRESH_CALLBACK);
4010
- await this.schedule(5, REFRESH_CALLBACK, resource);
4011
- return { resource, state: "degraded", reason, action: "rearmed" };
3897
+ return { resource, state: "degraded", reason, action: "none" };
4012
3898
  }
4013
3899
  }
4014
-
4015
- // No chain to be dead under the Workflow lifecycle (item 7).
4016
- if (!chained) return { resource, ...status, action: "none" };
4017
- const pending = await this.listSchedules(REFRESH_CALLBACK);
4018
- if (pending.length === 0) {
4019
- await this.schedule(5, REFRESH_CALLBACK, resource);
4020
- const reason = "alarm-missed: refresh chain was dead; re-armed by watchdog";
4021
- // An in-flight restore owns its own state; everything else is visibly
4022
- // degraded until the re-armed refresh succeeds.
4023
- if (status.state !== "restoring") await this.setResidentState("degraded", reason);
4024
- return { resource, state: "degraded", reason, action: "rearmed" };
4025
- }
4026
3900
  return { resource, ...status, action: "none" };
4027
3901
  }
4028
3902
 
@@ -4197,7 +4071,7 @@ export class ResidentDO extends Sandbox<Env> {
4197
4071
  }
4198
4072
  // From here the attach may hold the mirror lock through clone/install:
4199
4073
  // count it so a concurrent refresh-cycle reconcileImage never stops the
4200
- // container under it (and isIdle never parks the alarm mid-attach).
4074
+ // container under it (and isIdle never parks the cycle mid-attach).
4201
4075
  this.attachesInFlight++;
4202
4076
  try {
4203
4077
  return await this.attachThreadBody(threadKey, refHint, readonly, wantSha, resourceId, t0, record);
@@ -4213,16 +4087,15 @@ export class ResidentDO extends Sandbox<Env> {
4213
4087
  * on, named by step. A step that died of a full disk (docs/reference/specs/resident-repos.md item 54) also
4214
4088
  * flips the resident `degraded(disk-full: …)` — not serviceable, so the
4215
4089
  * next dispatch goes cold without attaching (the card names the disk, not
4216
- * `/etc/gitconfig.lock`) — and pulls the refresh cycle to now, where the
4217
- * recovery decision lives (an attach never stops the container itself: it
4218
- * is in flight). */
4219
- private async attachFailed(err: unknown, resource: string): Promise<ThreadErr> {
4090
+ * `/etc/gitconfig.lock`); the recovery decision lives in the next refresh
4091
+ * instance's entry gate (the cron's, within one bucket) — an attach never
4092
+ * stops the container itself: it is in flight. */
4093
+ private async attachFailed(err: unknown): Promise<ThreadErr> {
4220
4094
  if (!(err instanceof StepError)) return { error: `attach-failed: ${errMsg(err)}`, status: 500 };
4221
4095
  const failure = await this.classifyFailure(err.step, err.message);
4222
4096
  if (!failure.diskFull) return { error: `attach-failed at ${err.step}: ${err.message}`, status: 500 };
4223
4097
  console.log(`attach: ${failure.reason}`);
4224
4098
  await this.setResidentState("degraded", failure.reason);
4225
- await this.armRefresh(resource, DISK_FULL_REARM_S);
4226
4099
  return { error: `attach-failed: ${failure.reason}`, status: 500 };
4227
4100
  }
4228
4101
 
@@ -4294,7 +4167,6 @@ export class ResidentDO extends Sandbox<Env> {
4294
4167
  threadKey,
4295
4168
  refHint,
4296
4169
  wantSha,
4297
- resource,
4298
4170
  slug,
4299
4171
  t0,
4300
4172
  facts,
@@ -4316,7 +4188,6 @@ export class ResidentDO extends Sandbox<Env> {
4316
4188
  threadKey: string;
4317
4189
  refHint: string | null;
4318
4190
  wantSha: string | null;
4319
- resource: string;
4320
4191
  slug: string;
4321
4192
  t0: number;
4322
4193
  facts: RepoFacts;
@@ -4325,7 +4196,7 @@ export class ResidentDO extends Sandbox<Env> {
4325
4196
  mode: ReturnType<typeof planReadonlyAttach>;
4326
4197
  rollback: () => Promise<void>;
4327
4198
  }): Promise<AttachOk | ThreadErr> {
4328
- const { threadKey, refHint, wantSha, resource, slug, t0, facts, record, binding, mode, rollback } = input;
4199
+ const { threadKey, refHint, wantSha, slug, t0, facts, record, binding, mode, rollback } = input;
4329
4200
 
4330
4201
  // Command-level token mint — before the lock so mint latency
4331
4202
  // never holds the mutex, and failure never blocks the attach.
@@ -4414,7 +4285,7 @@ export class ResidentDO extends Sandbox<Env> {
4414
4285
  if (err instanceof StepError && err.step === "unknown-ref") {
4415
4286
  return { error: `unknown-ref: ${err.message}`, status: 400 };
4416
4287
  }
4417
- return this.attachFailed(err, resource);
4288
+ return this.attachFailed(err);
4418
4289
  }
4419
4290
 
4420
4291
  let deps: { deps: ThreadDepsMechanism; reconciled: boolean; depsKey?: string };
@@ -4449,7 +4320,7 @@ export class ResidentDO extends Sandbox<Env> {
4449
4320
  const s = await this.getStatus();
4450
4321
  return { error: errMsg(err), status: 503, state: s.state, reason: "mirror-busy" };
4451
4322
  }
4452
- return this.attachFailed(err, resource);
4323
+ return this.attachFailed(err);
4453
4324
  }
4454
4325
 
4455
4326
  // The prior write time and token expiry never survive an attach: a
@@ -4470,12 +4341,9 @@ export class ResidentDO extends Sandbox<Env> {
4470
4341
  ...(credentialsWrittenAt !== undefined ? { credentialsWrittenAt } : {}),
4471
4342
  ...(credentialTokenExpiresAtMs !== undefined ? { tokenExpiresAtMs: credentialTokenExpiresAtMs } : {}),
4472
4343
  } satisfies ThreadBinding);
4473
- if ((await this.listSchedules(SWEEP_CALLBACK)).length === 0) {
4474
- await this.schedule(SWEEP_INTERVAL_S, SWEEP_CALLBACK, resource);
4475
- }
4476
- // Item 55: the tree is on disk now — measure it (deferred; the `du` stays
4477
- // off this hot path) so the next admission projects from current parts.
4478
- await this.scheduleDiskMeasure(resource);
4344
+ // Item 55: the tree is on disk now; the next refresh instance's `measure`
4345
+ // step counts it — the admission's free-space term is a live `df`, and the
4346
+ // per-part projection it reads from the sample moves only with a cycle.
4479
4347
 
4480
4348
  return {
4481
4349
  workspace: binding.worktreePath,
@@ -5489,9 +5357,8 @@ export class ResidentDO extends Sandbox<Env> {
5489
5357
  if (!(await this.evictBinding(current, activeNow, `detach`, "detach"))) {
5490
5358
  return { released: false, reason: "re-attached during eviction — kept", user };
5491
5359
  }
5492
- // Item 55: the tree is gone — re-measure (deferred) so the gauge and the
5493
- // next admission see the space back.
5494
- await this.scheduleDiskMeasure((await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "");
5360
+ // Item 55: the tree is gone; the gauge catches up at the next refresh
5361
+ // instance's `measure` step, and the admission's `df` sees the space now.
5495
5362
  return { released: true, user };
5496
5363
  }
5497
5364
 
@@ -5578,84 +5445,64 @@ export class ResidentDO extends Sandbox<Env> {
5578
5445
  /** A refresh cycle past its idle/reconcile gates (fetching, rebuilding, snapshotting). */
5579
5446
  private refreshesInFlight = 0;
5580
5447
 
5581
- /** Hourly inactivity sweep (schedule: onWorktreeSweep). Removes worktrees
5582
- * whose binding is idle past the TTL, releases the user to the pool, and
5583
- * KEEPS the binding record marked evicted. Never wakes a slept
5584
- * container just to delete files a sleep already destroyed. */
5585
- /** True while onWorktreeSweep is executing (DO memory; a restart clears it
5586
- * together with the in-flight sweep). The watchdog's re-arm checks it so
5587
- * two sweep chains can never be armed by construction. */
5588
- private sweepInFlight = false;
5589
-
5590
- async onWorktreeSweep(payload: string): Promise<{ evicted: string[]; kept: number }> {
5591
- const resource = payload || ((await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "");
5448
+ /** The worktree inactivity sweep — every refresh instance's `sweep` step
5449
+ * (item 23) and the `sweep-now` debug op. Removes worktrees whose binding
5450
+ * is idle past the TTL, releases the user to the pool, and KEEPS the
5451
+ * binding record marked evicted. Never wakes a slept container just to
5452
+ * delete files a sleep already destroyed. */
5453
+ async sweepWorktrees(resource: string): Promise<{ evicted: string[]; kept: number }> {
5592
5454
  const evicted: string[] = [];
5593
5455
  let kept = 0;
5594
- this.sweepInFlight = true;
5595
- try {
5596
- const record = await this.registry()
5597
- .getRecord(resource)
5598
- .catch(() => null);
5599
- const ttlDays = record?.worktreeTtlDays ?? WORKTREE_TTL_DAYS_DEFAULT;
5600
- const cutoff = systemClock() - ttlDays * 86_400_000;
5601
- const all = await this.ctx.storage.list<ThreadBinding>({ prefix: THREAD_KEY_PREFIX });
5602
- const active = await this.isRuntimeActive().catch(() => false);
5603
- const idleCutoff = systemClock() - CLEAN_IDLE_RELEASE_S * 1000;
5604
- for (const binding of all.values()) {
5605
- if (binding.evicted || !binding.user) continue;
5606
- const last = Date.parse(binding.lastAttachAt);
5607
- if (last >= cutoff) {
5608
- // Not past the TTL. Still release it if it has been idle for an hour,
5609
- // nothing is running on it, and the tree is provably clean — the run
5610
- // that used it is over and there is nothing to preserve. A slept
5611
- // container has NO tree any more (sleep destroys the disk), so an
5612
- // idle binding on an inactive runtime is releasable outright: there is
5613
- // nothing left to protect, only a pool user to give back. (Keeping
5614
- // them would leave idle bindings on a sleeping resident until the
5615
- // 7-day TTL.)
5616
- const busy = this.threadOpsInFlight.get(binding.threadKey) ?? 0;
5617
- const cleanIdle =
5618
- last < idleCutoff && busy === 0 && (!active || (await this.worktreeCleanliness(binding)).clean);
5619
- // Re-read right before removal: the clean check awaited (the DO
5620
- // yields), so an exec that arrived meanwhile would otherwise have
5621
- // its tree removed under it — same guard as detachThread.
5622
- const busyNow = this.threadOpsInFlight.get(binding.threadKey) ?? 0;
5623
- if (!cleanIdle || busyNow > 0) {
5624
- kept++;
5625
- continue;
5626
- }
5627
- }
5628
- // Re-read the binding too: a re-attach that completed inside the
5629
- // clean-check await bumped lastAttachAt and rebuilt the tree — evicting
5630
- // from this loop's stale snapshot would rm the fresh tree.
5631
- const current = await this.ctx.storage.get<ThreadBinding>(threadBindingKey(binding.threadKey));
5632
- if (!current || current.evicted || current.lastAttachAt !== binding.lastAttachAt) {
5456
+ const record = await this.registry()
5457
+ .getRecord(resource)
5458
+ .catch(() => null);
5459
+ const ttlDays = record?.worktreeTtlDays ?? WORKTREE_TTL_DAYS_DEFAULT;
5460
+ const cutoff = systemClock() - ttlDays * 86_400_000;
5461
+ const all = await this.ctx.storage.list<ThreadBinding>({ prefix: THREAD_KEY_PREFIX });
5462
+ const active = await this.isRuntimeActive().catch(() => false);
5463
+ const idleCutoff = systemClock() - CLEAN_IDLE_RELEASE_S * 1000;
5464
+ for (const binding of all.values()) {
5465
+ if (binding.evicted || !binding.user) continue;
5466
+ const last = Date.parse(binding.lastAttachAt);
5467
+ if (last >= cutoff) {
5468
+ // Not past the TTL. Still release it if it has been idle for an hour,
5469
+ // nothing is running on it, and the tree is provably clean — the run
5470
+ // that used it is over and there is nothing to preserve. A slept
5471
+ // container has NO tree any more (sleep destroys the disk), so an
5472
+ // idle binding on an inactive runtime is releasable outright: there is
5473
+ // nothing left to protect, only a pool user to give back. (Keeping
5474
+ // them would leave idle bindings on a sleeping resident until the
5475
+ // 7-day TTL.)
5476
+ const busy = this.threadOpsInFlight.get(binding.threadKey) ?? 0;
5477
+ const cleanIdle =
5478
+ last < idleCutoff && busy === 0 && (!active || (await this.worktreeCleanliness(binding)).clean);
5479
+ // Re-read right before removal: the clean check awaited (the DO
5480
+ // yields), so an exec that arrived meanwhile would otherwise have
5481
+ // its tree removed under it — same guard as detachThread.
5482
+ const busyNow = this.threadOpsInFlight.get(binding.threadKey) ?? 0;
5483
+ if (!cleanIdle || busyNow > 0) {
5633
5484
  kept++;
5634
5485
  continue;
5635
5486
  }
5636
- // `active` is re-read per binding: the container can wake mid-sweep (an
5637
- // attach), and an eviction decided on a stale "inactive" would skip the
5638
- // rm and orphan a real tree.
5639
- const activeNow = await this.isRuntimeActive().catch(() => false);
5640
- if (
5641
- await this.evictBinding(
5642
- current,
5643
- activeNow,
5644
- `worktree-sweep ${resource}`,
5645
- last >= cutoff ? "clean-idle" : "ttl",
5646
- )
5647
- )
5648
- evicted.push(binding.threadKey);
5649
- else kept++;
5650
5487
  }
5651
- } finally {
5652
- const all = await this.ctx.storage.list<ThreadBinding>({ prefix: THREAD_KEY_PREFIX });
5653
- const live = [...all.values()].some((b) => !b.evicted && b.user);
5654
- this.deleteSchedules(SWEEP_CALLBACK);
5655
- if (live) await this.schedule(SWEEP_INTERVAL_S, SWEEP_CALLBACK, resource);
5656
- this.sweepInFlight = false;
5488
+ // Re-read the binding too: a re-attach that completed inside the
5489
+ // clean-check await bumped lastAttachAt and rebuilt the tree — evicting
5490
+ // from this loop's stale snapshot would rm the fresh tree.
5491
+ const current = await this.ctx.storage.get<ThreadBinding>(threadBindingKey(binding.threadKey));
5492
+ if (!current || current.evicted || current.lastAttachAt !== binding.lastAttachAt) {
5493
+ kept++;
5494
+ continue;
5495
+ }
5496
+ // `active` is re-read per binding: the container can wake mid-sweep (an
5497
+ // attach), and an eviction decided on a stale "inactive" would skip the
5498
+ // rm and orphan a real tree.
5499
+ const activeNow = await this.isRuntimeActive().catch(() => false);
5500
+ if (
5501
+ await this.evictBinding(current, activeNow, `worktree-sweep ${resource}`, last >= cutoff ? "clean-idle" : "ttl")
5502
+ )
5503
+ evicted.push(binding.threadKey);
5504
+ else kept++;
5657
5505
  }
5658
- if (evicted.length > 0) await this.scheduleDiskMeasure(resource); // item 55
5659
5506
  return { evicted, kept };
5660
5507
  }
5661
5508
 
@@ -5870,9 +5717,9 @@ export class ResidentDO extends Sandbox<Env> {
5870
5717
  return { ok: true, lastAttachAt };
5871
5718
  }
5872
5719
 
5873
- /** Debug: run the sweep pass now (the exact scheduled function). */
5720
+ /** Debug: run the sweep pass now (the exact function the `sweep` step runs). */
5874
5721
  async debugSweepNow(): Promise<{ evicted: string[]; kept: number }> {
5875
- return this.onWorktreeSweep("");
5722
+ return this.sweepWorktrees((await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "");
5876
5723
  }
5877
5724
 
5878
5725
  // -- event-triggered reclamation ---------------------------------------------
@@ -5908,7 +5755,7 @@ export class ResidentDO extends Sandbox<Env> {
5908
5755
  /** Reclaim worktrees whose ref is FINISHED: the branch vanished from the
5909
5756
  * mirror (the refresh cycle's `fetch --prune` just ran) or its PR was
5910
5757
  * merged/closed. Runs inside the refresh cycle — a poll on the existing
5911
- * alarm, since the GitHub App has webhooks off — and via /debug
5758
+ * cadence, since the GitHub App has webhooks off — and via /debug
5912
5759
  * reclaim-now. Never touches the default branch, a busy thread, or a dirty
5913
5760
  * tree (reclaimDecision); every keep is named. The eviction itself is the
5914
5761
  * sweep's `evictBinding` with the same re-read guards. */
@@ -6098,8 +5945,7 @@ export class ResidentDO extends Sandbox<Env> {
6098
5945
  instance: null,
6099
5946
  skipped: null,
6100
5947
  };
6101
- const [refresh, provisionRun, provisionDeadline, bindings] = await Promise.all([
6102
- this.listSchedules(REFRESH_CALLBACK),
5948
+ const [provisionRun, provisionDeadline, bindings] = await Promise.all([
6103
5949
  this.listSchedules(PROVISION_RUN_CALLBACK),
6104
5950
  this.listSchedules(PROVISIONING_CALLBACK),
6105
5951
  this.ctx.storage.list<ThreadBinding>({ prefix: THREAD_KEY_PREFIX }),
@@ -6144,8 +5990,8 @@ export class ResidentDO extends Sandbox<Env> {
6144
5990
  checkoutBackupId: snap.checkout.id,
6145
5991
  }
6146
5992
  : null,
5993
+ // The provisioning schedules, the one timer a resident has (item 3).
6147
5994
  schedules: {
6148
- refresh: refresh.length,
6149
5995
  provisionRun: provisionRun.length,
6150
5996
  provisionDeadline: provisionDeadline.length,
6151
5997
  },
@@ -6160,9 +6006,8 @@ export class ResidentDO extends Sandbox<Env> {
6160
6006
  map.get(inFlightKey("refresh")) as Lease | undefined,
6161
6007
  map.get(inFlightKey("hydration")) as Lease | undefined,
6162
6008
  ),
6163
- // Item 7: which scheduler drives the refresh cycle, and — on the
6164
- // Workflow lifecycle — the instance the cron last created with the step
6165
- // it last reported, and the last bucket the cron skipped.
6009
+ // Item 7: the lifecycle row (`workflow`), the instance last created with
6010
+ // the step it last reported, and the last bucket the cron skipped.
6166
6011
  lifecycle: lifecycleOf(map.get(LIFECYCLE_KEY)),
6167
6012
  refresh: {
6168
6013
  instance: refreshRow.instance
@@ -6183,32 +6028,15 @@ export class ResidentDO extends Sandbox<Env> {
6183
6028
 
6184
6029
  // -- debug surface (admin-scoped via POST /debug; used by live validation) ---
6185
6030
 
6031
+ /** The pending schedule rows: provisioning's two, the one timer a resident
6032
+ * has — a settled resident answers both empty (lifecycle.test.ts holds that
6033
+ * nothing else is ever scheduled). */
6186
6034
  async debugSchedules(): Promise<Record<string, unknown>> {
6187
- const [refresh, provisionRun, provisionDeadline, sweep] = await Promise.all([
6188
- this.listSchedules(REFRESH_CALLBACK),
6035
+ const [provisionRun, provisionDeadline] = await Promise.all([
6189
6036
  this.listSchedules(PROVISION_RUN_CALLBACK),
6190
6037
  this.listSchedules(PROVISIONING_CALLBACK),
6191
- this.listSchedules(SWEEP_CALLBACK),
6192
6038
  ]);
6193
- return { refresh, provisionRun, provisionDeadline, sweep };
6194
- }
6195
-
6196
- /** Kill the refresh chain (simulates a dead alarm chain for watchdog tests). */
6197
- async debugKillRefresh(): Promise<{ killed: boolean; remaining: number }> {
6198
- this.deleteSchedules(REFRESH_CALLBACK);
6199
- return { killed: true, remaining: (await this.listSchedules(REFRESH_CALLBACK)).length };
6200
- }
6201
-
6202
- /** Pull the next refresh forward to ~1s from now. A resident on the Workflow
6203
- * lifecycle has no chain to pull: its next cycle is the instance the next
6204
- * cron firing creates (`run-watchdog` runs that pass on demand). */
6205
- async debugRefreshNow(): Promise<{ scheduled: boolean; lifecycle: ResidentLifecycle }> {
6206
- const lifecycle = await this.getLifecycle();
6207
- if (lifecycle === "workflow") return { scheduled: false, lifecycle };
6208
- const resource = (await this.ctx.storage.get<string>(RESOURCE_KEY)) ?? "";
6209
- this.deleteSchedules(REFRESH_CALLBACK);
6210
- await this.schedule(1, REFRESH_CALLBACK, resource);
6211
- return { scheduled: true, lifecycle };
6039
+ return { provisionRun, provisionDeadline };
6212
6040
  }
6213
6041
 
6214
6042
  /** Fault injection for the watchdog's stuck-onboarding path: re-persist
@@ -6234,23 +6062,21 @@ export class ResidentDO extends Sandbox<Env> {
6234
6062
  }
6235
6063
 
6236
6064
  /** Fault injection for the watchdog's auto-rebuild path: persist
6237
- * `down` with a rehydration-flavored reason and stop the refresh chain
6238
- * (mirroring what a real goDown does), so repeated watchdog passes can
6239
- * strike it up to the auto-rebuild without corrupting real R2 objects.
6240
- * Test-only semantics; admin scope. */
6065
+ * `down` with a rehydration-flavored reason (what a real goDown does), so
6066
+ * repeated watchdog passes can strike it up to the auto-rebuild without
6067
+ * corrupting real R2 objects. Test-only semantics; admin scope. */
6241
6068
  async debugForceDown(reason: string): Promise<ResidentStatus> {
6242
6069
  await this.setResidentState("down", reason);
6243
- this.deleteSchedules(REFRESH_CALLBACK);
6244
6070
  return this.getStatus();
6245
6071
  }
6246
6072
 
6247
6073
  /** Rebuild: the down→onboarding escape hatch — discard the recorded
6248
6074
  * snapshots (R2 objects included) and reprovision from scratch through the
6249
- * ordinary alarm-driven pipeline, reusing the registry record's command
6075
+ * ordinary provisioning pipeline, reusing the registry record's command
6250
6076
  * table/ref/budget. `dryRun` returns the same itemized plan WITHOUT
6251
6077
  * executing: nothing deleted, no state change, schedules untouched.
6252
6078
  * Refused while the engine owns the state (onboarding/refreshing/
6253
- * restoring) — two engine chains must never race the same disk. */
6079
+ * restoring) — two cycles must never race the same disk. */
6254
6080
  async rebuild(
6255
6081
  resource: string,
6256
6082
  defaultRef: string,
@@ -6338,18 +6164,16 @@ export class ResidentDO extends Sandbox<Env> {
6338
6164
  threadBindings: number;
6339
6165
  }> {
6340
6166
  const status = await this.getStatus();
6341
- const [prov, run, refresh, sweep] = await Promise.all([
6167
+ const [prov, run] = await Promise.all([
6342
6168
  this.listSchedules(PROVISIONING_CALLBACK),
6343
6169
  this.listSchedules(PROVISION_RUN_CALLBACK),
6344
- this.listSchedules(REFRESH_CALLBACK),
6345
- this.listSchedules(SWEEP_CALLBACK),
6346
6170
  ]);
6347
6171
  const snap = await this.ctx.storage.get<SnapshotRecord>(SNAPSHOT_KEY);
6348
6172
  const ids = snap ? [snap.mirror.id, snap.checkout.id] : [];
6349
6173
  const bindings = await this.ctx.storage.list<ThreadBinding>({ prefix: THREAD_KEY_PREFIX });
6350
6174
  return {
6351
6175
  ...status,
6352
- schedules: prov.length + run.length + refresh.length + sweep.length,
6176
+ schedules: prov.length + run.length,
6353
6177
  snapshotBackupIds: ids,
6354
6178
  backupObjects: await this.countBackupObjects(ids),
6355
6179
  threadBindings: bindings.size,
@@ -6373,8 +6197,6 @@ export class ResidentDO extends Sandbox<Env> {
6373
6197
  let backupObjectsDeleted = 0;
6374
6198
  this.deleteSchedules(PROVISIONING_CALLBACK);
6375
6199
  this.deleteSchedules(PROVISION_RUN_CALLBACK);
6376
- this.deleteSchedules(REFRESH_CALLBACK);
6377
- this.deleteSchedules(SWEEP_CALLBACK);
6378
6200
  const snap = await this.ctx.storage.get<SnapshotRecord>(SNAPSHOT_KEY);
6379
6201
  if (snap) {
6380
6202
  try {
@@ -6751,11 +6573,13 @@ export default {
6751
6573
  return res;
6752
6574
  },
6753
6575
 
6754
- /** Watchdog cron: one sparse pass that re-arms dead refresh chains
6755
- * (marking degraded(alarm-missed)) and times out stuck onboarding. Cadence
6756
- * invariant: this cron (every 10 minutes) stays SHORTER than SLEEP_AFTER
6757
- * ("20m"). It reads DO storage/schedules only — containers are started by
6758
- * the re-armed refresh alarms, not by the watchdog itself.
6576
+ /** Watchdog cron: one sparse pass that creates each resident's due refresh
6577
+ * instance (item 7), names a stale mid-flight marker by its instance, and
6578
+ * times out stuck onboarding. Cadence invariant: this cron (every 10
6579
+ * minutes) stays SHORTER than SLEEP_AFTER ("20m") — the instance it
6580
+ * creates is the keep-warm. It reads DO storage and the engine's instance
6581
+ * status only — containers are started by the instances' steps, not by the
6582
+ * watchdog itself.
6759
6583
  *
6760
6584
  * The cron is the `resident` entry of the schedule registry
6761
6585
  * (src/core/schedules.ts — a unit test keeps wrangler.jsonc equal to it);
@@ -7102,7 +6926,7 @@ async function handleReconfigure(env: Env, body: Record<string, unknown>): Promi
7102
6926
  * reusing the registry record (command table, ref, budget) as-is; the cap
7103
6927
  * slot and thread bindings are untouched. `dryRun:true` answers 200 with the
7104
6928
  * itemized plan and executes nothing; a real rebuild answers 202 like
7105
- * onboard (the transition is alarm-driven). */
6929
+ * onboard (the transition is provisioning's schedule). */
7106
6930
  async function handleRebuild(env: Env, body: Record<string, unknown>): Promise<Response> {
7107
6931
  const resource = parseResource(body.resource);
7108
6932
  if ("error" in resource) return json({ error: resource.error }, 400);
@@ -7464,8 +7288,8 @@ function streamOp(pending: Promise<Awaited<ReturnType<ResidentDO["runOp"]>>>): R
7464
7288
  );
7465
7289
  }
7466
7290
 
7467
- /** Admin diagnostic surface, used by the live validation of the freshness engine (kill-refresh /
7468
- * stop-container simulate dead chains and platform sleeps; mint-token proves
7291
+ /** Admin diagnostic surface, used by the live validation of the freshness engine (refresh-now
7292
+ * creates this bucket's instance on demand, stop-container simulates a platform sleep; mint-token proves
7469
7293
  * the command-level mint failure shape without exposing token material).
7470
7294
  * Side-effect-explicit; every op is admin-scope except the pure reads
7471
7295
  * info/schedules/threads, which the read scope may also run. */
@@ -7513,10 +7337,14 @@ async function handleDebug(env: Env, body: Record<string, unknown>): Promise<Res
7513
7337
  return json(await stub.getResidentInfo());
7514
7338
  case "schedules":
7515
7339
  return json(await stub.debugSchedules());
7516
- case "kill-refresh":
7517
- return json(await stub.debugKillRefresh());
7518
7340
  case "refresh-now":
7519
- return json(await stub.debugRefreshNow());
7341
+ // Item 13: this bucket's refresh instance, created now — `duplicate` when
7342
+ // the cron already served the bucket, `skipped` beside a live cycle.
7343
+ return json({
7344
+ op,
7345
+ resource: resource.resource,
7346
+ ...(await createRefreshInstanceNow(env, stub, resource.resource)),
7347
+ });
7520
7348
  case "stop-container":
7521
7349
  return json(await stub.debugStopContainer());
7522
7350
  case "force-onboarding":
@@ -7555,16 +7383,16 @@ async function handleDebug(env: Env, body: Record<string, unknown>): Promise<Res
7555
7383
  return json(await stub.debugBackdateThread(threadKey.threadKey, days.value));
7556
7384
  }
7557
7385
  case "lifecycle": {
7558
- // Item 7: which scheduler drives this resident's refresh cycle. Admin
7559
- // scope, one resident at a time; the default for every resident is `alarm`.
7386
+ // Item 7: the lifecycle row. `workflow` is the one scheduler, so the op
7387
+ // only rewrites a stale `alarm` value the flagged rollout left behind.
7560
7388
  const mode = parseLifecycle(body.mode);
7561
- if (!mode) return json({ error: 'mode must be "alarm" or "workflow"' }, 400);
7389
+ if (!mode) return json({ error: 'mode must be "workflow" — the alarm chain no longer exists' }, 400);
7562
7390
  return json({ op, resource: resource.resource, ...(await stub.setLifecycle(mode)) });
7563
7391
  }
7564
7392
  default:
7565
7393
  return json(
7566
7394
  {
7567
- error: `unknown op ${JSON.stringify(op)} (ops: info, schedules, kill-refresh, refresh-now, stop-container, force-onboarding, force-down, mint-token, run-watchdog, set-test-overrides, threads, sweep-now, reclaim-now, measure-disk, purge-bindings, backdate-thread, lifecycle)`,
7395
+ error: `unknown op ${JSON.stringify(op)} (ops: info, schedules, refresh-now, stop-container, force-onboarding, force-down, mint-token, run-watchdog, set-test-overrides, threads, sweep-now, reclaim-now, measure-disk, purge-bindings, backdate-thread, lifecycle)`,
7568
7396
  },
7569
7397
  400,
7570
7398
  );
@@ -7575,8 +7403,8 @@ async function handleDebug(env: Env, body: Record<string, unknown>): Promise<Res
7575
7403
  * handler and the /debug run-watchdog op. Each check targets a different DO,
7576
7404
  * so they run concurrently; a failing one becomes its own {error} entry
7577
7405
  * without touching its neighbors, and the results follow the registry list.
7578
- * For a resident on the Workflow lifecycle the pass also creates the refresh
7579
- * instance the bucket is due (item 7). */
7406
+ * The pass also creates each resident's refresh instance when its bucket is
7407
+ * due (item 7). */
7580
7408
  async function runWatchdog(env: Env, parent?: TraceSpan): Promise<WatchdogSummary> {
7581
7409
  const registry = registryStub(env);
7582
7410
  const residents = await registry.list();
@@ -7608,7 +7436,6 @@ async function runWatchdog(env: Env, parent?: TraceSpan): Promise<WatchdogSummar
7608
7436
  reason: s.value.reason,
7609
7437
  action: s.value.action,
7610
7438
  disk: s.value.disk,
7611
- lifecycle: s.value.refresh.lifecycle,
7612
7439
  instance: s.value.instance,
7613
7440
  }
7614
7441
  : { resource: record.resource, error: errMsg(s.reason) };