@coreplane/switchboard 1.201.0 → 1.202.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. package/README.md +1 -1
  2. package/bin/switchboard.js +26 -0
  3. package/dist/assets/deploy/cloudflare-resident/refresh.ts +238 -0
  4. package/dist/assets/deploy/cloudflare-resident/shared.ts +80 -0
  5. package/dist/assets/deploy/cloudflare-resident/tsconfig.json +1 -1
  6. package/dist/assets/deploy/cloudflare-resident/worker.ts +67 -285
  7. package/dist/assets/deploy/cloudflare-resident/wrangler.template.jsonc +2 -1
  8. package/dist/assets/package-lock.json +481 -4
  9. package/dist/assets/package.json +1 -1
  10. package/dist/assets/project.json +1 -1
  11. package/dist/assets/source.json +3 -3
  12. package/dist/assets/src/core/authz/policy.ts +3 -0
  13. package/dist/assets/src/execution/residentRefresh.ts +29 -0
  14. package/dist/assets/web/dist/.vite/manifest.json +43 -43
  15. package/dist/assets/web/dist/assets/{AppShell-CjDyTiip.js → AppShell-CJLPO_aI.js} +1 -1
  16. package/dist/assets/web/dist/assets/{CostsPage-CMAh3G74.js → CostsPage-C6ZvnAmH.js} +1 -1
  17. package/dist/assets/web/dist/assets/{NotFoundPage-CbEKm0tY.js → NotFoundPage-BdLcY60r.js} +1 -1
  18. package/dist/assets/web/dist/assets/{ResidentDetailPage-CcfmOAIP.js → ResidentDetailPage-hXWrr8T6.js} +1 -1
  19. package/dist/assets/web/dist/assets/{ResidentsIndexPage-9Z5jcteM.js → ResidentsIndexPage-YLu4JwxJ.js} +1 -1
  20. package/dist/assets/web/dist/assets/{RunRoutePage-Cdimh4e0.js → RunRoutePage-BDhHRJVO.js} +1 -1
  21. package/dist/assets/web/dist/assets/{RunsIndexPage-CvgDXrO4.js → RunsIndexPage-D7zZ5a0l.js} +1 -1
  22. package/dist/assets/web/dist/assets/{RunsTabs-rEjVQG6H.js → RunsTabs-BOUlSa2W.js} +1 -1
  23. package/dist/assets/web/dist/assets/{ScheduledPage-C5bG9qvd.js → ScheduledPage-BciOQcC-.js} +1 -1
  24. package/dist/assets/web/dist/assets/{StatusDot-D_IPK3WP.js → StatusDot-CTDLBv92.js} +1 -1
  25. package/dist/assets/web/dist/assets/{Tooltip-JlZp3OVC.js → Tooltip-DD2v9Gxx.js} +1 -1
  26. package/dist/assets/web/dist/assets/{favicon-DZDjc8ab.js → favicon-C4Q8cRsP.js} +1 -1
  27. package/dist/assets/web/dist/assets/{main-BKKzFAnK.js → main-DrSlUcMg.js} +2 -2
  28. package/dist/cli.js +629 -207
  29. package/package.json +3 -2
@@ -65,7 +65,6 @@
65
65
  // (untrusted repo code) run unprivileged (worker1) and token-free
66
66
  // (docs/decisions/0009-residents-second-credential-domain.md).
67
67
  import {
68
- getSandbox,
69
68
  isDurableObjectCodeUpdateReset,
70
69
  OperationInterruptedError,
71
70
  ProcessWaitTimeoutError,
@@ -77,7 +76,7 @@ import {
77
76
  import { AsyncLocalStorage } from "node:async_hooks";
78
77
  import type { DirectoryBackup, SandboxCommand } from "@cloudflare/sandbox";
79
78
  import { createExtensionProcessSandbox } from "@cloudflare/sandbox/extensions";
80
- import { DurableObject, WorkflowEntrypoint, type WorkflowEvent, type WorkflowStep } from "cloudflare:workers";
79
+ import { DurableObject } from "cloudflare:workers";
81
80
  import { BASH_TIMEOUT_MAX_MS, BASH_TIMEOUT_MS, clampBashTimeout } from "../../src/execution/bashTimeout.js";
82
81
  import { selectBindingsToPurge } from "../../src/execution/bindingPurge.js";
83
82
  import { busyAfterKillReason, planForceDetach } from "../../src/execution/residentDetach.js";
@@ -130,6 +129,7 @@ import {
130
129
  import {
131
130
  checkoutUpdateCommand,
132
131
  classifyRefreshFailure,
132
+ restoreFailureDisposition,
133
133
  killStaleBuildProcessesCommand,
134
134
  nextRefreshDelayS,
135
135
  planRefresh,
@@ -147,12 +147,8 @@ import {
147
147
  type RefreshOutcome,
148
148
  } from "../../src/execution/residentRefresh.js";
149
149
  import {
150
- REFRESH_STEP_RETRIES,
151
150
  lifecycleOf,
152
151
  parseLifecycle,
153
- refreshInstanceId,
154
- shouldCreateRefreshInstance,
155
- stepTimeoutMs,
156
152
  type RefreshRow,
157
153
  type ResidentLifecycle,
158
154
  } from "../../src/execution/residentInstanceId.js";
@@ -221,8 +217,7 @@ import { graftResidentSteps } from "../../src/execution/residentTrace.js";
221
217
  import type { SpanAttrs } from "../../src/core/trace/attrs.js";
222
218
  import type { Span as TraceSpan } from "../../src/core/trace/types.js";
223
219
  import { systemClock } from "../../src/core/trace/clock.js";
224
- import { createTracer } from "../../src/core/trace/tracer.js";
225
- import { startAdoptedRoot, workerLogSink } from "../../src/core/trace/workerTrace.js";
220
+ import { startAdoptedRoot } from "../../src/core/trace/workerTrace.js";
226
221
  import { backupTransferMode } from "../../src/execution/residentBackupTransfer.js";
227
222
  import {
228
223
  extractRestoreScript,
@@ -255,6 +250,27 @@ import {
255
250
  planDepsMaterialization,
256
251
  } from "../../src/execution/residentDepsStore.js";
257
252
  import { buildId, injectedBuildStamp } from "../../src/deploy/buildStamp.js";
253
+ import { createRefreshInstance, type RefreshInstanceParams } from "./refresh";
254
+ import {
255
+ DEFAULT_EXEC_TIMEOUT_MS,
256
+ DEPS_STEP_OVERHEAD_MS,
257
+ errMsg,
258
+ GIT_NETWORK_TIMEOUT_MS,
259
+ IDLE_REFRESH_INTERVAL_S,
260
+ R2_TRANSFER_TIMEOUT_MS,
261
+ REFRESH_BUILD_TIMEOUT_MS,
262
+ REFRESH_INSTALL_TIMEOUT_MS,
263
+ REFRESH_INTERVAL_S,
264
+ registryStub,
265
+ residentStub,
266
+ tracer,
267
+ traceSinks,
268
+ } from "./shared";
269
+
270
+ /** The refresh cycle's Workflow entrypoint is declared in refresh.ts; the
271
+ * Workflows binding resolves its `class_name` against this module
272
+ * (wrangler.template.jsonc), so the entry exports it under that name. */
273
+ export { ResidentRefresh } from "./refresh";
258
274
 
259
275
  /** The commit this bundle was built from, injected by the deploy
260
276
  * (`deploy/bin/build-stamp.mjs`; `unknown` when nobody stamped it). Answered
@@ -270,9 +286,8 @@ const BUILD_ID = buildId(BUILD);
270
286
  // (/attach, /exec, /op) are rooted inside the DO where the work is, their
271
287
  // collector's steps as `resident.<step>` children; every other authenticated
272
288
  // route is a `resident.fetch` root at the edge. Each joins the bot's trace when
273
- // the request carried one. A `slow` log sink whose filter drops a refusal's line.
274
- const tracer = createTracer({ clock: systemClock });
275
- const traceSinks = [workerLogSink((line) => console.log(line))];
289
+ // the request carried one. The tracer and its log sink are shared.ts's: the
290
+ // refresh instance's root (refresh.ts) starts from the same pair.
276
291
  const STREAMED_ROUTES: ReadonlySet<string> = new Set(["/attach", "/exec", "/op"]);
277
292
 
278
293
  /** One request as the resident's own root: started at its t0, joining the
@@ -296,20 +311,14 @@ function refusalOutcome(err: ThreadErr): string {
296
311
  return err.needs ? `needs_${err.needs}` : "error";
297
312
  }
298
313
 
299
- /** What a refresh instance is created with: the resident it runs for. Every
300
- * other input is read from the resident's rows at each step, never carried. */
301
- interface RefreshInstanceParams {
302
- resource: string;
303
- }
304
-
305
- interface Env {
314
+ export interface Env {
306
315
  RESIDENT: DurableObjectNamespace<ResidentDO>;
307
316
  REGISTRY: DurableObjectNamespace<ResidentRegistryDO>;
308
317
  BACKUP_BUCKET: R2Bucket;
309
318
  /** The refresh cycle as a Workflow instance (docs/reference/specs/resident-repos.md
310
- * item 7): `ResidentRefresh` below. The watchdog cron creates one per
311
- * resident whose row says `lifecycle: workflow`; an `alarm` resident (the
312
- * default) never has one. */
319
+ * item 7): `ResidentRefresh` in refresh.ts, re-exported above. The watchdog
320
+ * cron creates one per resident whose row says `lifecycle: workflow`; an
321
+ * `alarm` resident (the default) never has one. */
313
322
  RESIDENT_REFRESH: Workflow<RefreshInstanceParams>;
314
323
  // Presigned snapshot transfers (docs/reference/specs/resident-repos.md item 61): with all
315
324
  // four present the container moves archive bytes itself over presigned R2
@@ -354,19 +363,6 @@ interface Env {
354
363
  // `evictColdest:true` makes room (docs/reference/specs/resident-repos.md item 46).
355
364
  const RESIDENT_CAP = 6;
356
365
 
357
- /** Container sleep window, passed to every getSandbox() for ResidentDO.
358
- * Invariant: REFRESH_INTERVAL_S and the watchdog cron (wrangler.jsonc,
359
- * every 10 minutes) MUST both stay SHORTER than this window, so a healthy
360
- * resident is re-warmed before the platform can sleep it. Bump together. */
361
- const SLEEP_AFTER = "20m";
362
-
363
- /** Refresh alarm cadence (seconds). Each resident DO self-reschedules this
364
- * alarm (per-resident alarms own freshness; the sparse cron is only the
365
- * watchdog); it doubles as the keep-warm heartbeat, so it must stay below
366
- * SLEEP_AFTER. Matches the watchdog cron so a killed chain is re-armed
367
- * within one refresh interval. */
368
- const REFRESH_INTERVAL_S = 600;
369
-
370
366
  /** R2 lifetime of snapshot objects. We delete replaced/offboarded snapshots
371
367
  * explicitly (see deleteBackupObjects); the TTL is a leak backstop, and it
372
368
  * must be long — a quiet repo's current snapshot may go unreplaced for
@@ -379,46 +375,16 @@ const DEFAULT_PROVISIONING_TIMEOUT_MS = 5 * 60_000;
379
375
  const MIN_PROVISIONING_TIMEOUT_MS = 10_000;
380
376
  const MAX_PROVISIONING_TIMEOUT_MS = 30 * 60_000;
381
377
 
382
- /** Exec budgets. The DO alarm handler has a ~15-minute platform wall clock;
383
- * every schedule callback's step budgets are chosen to fit under it. */
384
- const DEFAULT_EXEC_TIMEOUT_MS = 60_000;
385
- const GIT_NETWORK_TIMEOUT_MS = 5 * 60_000;
386
- const REFRESH_BUILD_TIMEOUT_MS = 5 * 60_000;
387
- /** The refresh install budget. Twice the build's: a full `npm install` of the
388
- * switchboard lockfile takes ~4 min on the resident's 1 vCPU when nothing
389
- * else runs, and thread runs (tests, a review's greps) share that vCPU —
390
- * under load it crosses 5 min several cycles in a row while the default
391
- * branch keeps moving. A timed-out install is worse than a slow one: the cycle's whole
392
- * budget is spent and the checkout is left without deps, so the next cycle
393
- * starts the same install over. The cycle runs in the background (runs keep
394
- * attaching to the last snapshot); the cost of a longer budget is a longer
395
- * mirror-lock window, bounded well inside STALE_MIDFLIGHT_MS. */
396
- const REFRESH_INSTALL_TIMEOUT_MS = 10 * 60_000;
397
378
  /** After `output()`'s wait gives up on a process the supervisor should have
398
379
  * killed at `timeout`, how long to wait for the exit status of OUR kill
399
380
  * before reporting the step without one. */
400
381
  const KILL_EXIT_WAIT_MS = 10_000;
401
- /** Budget per R2 SNAPSHOT upload. The SDK's createBackup
402
- * accepts no timeout or AbortSignal, so each call is raced against this
403
- * (withTimeout): a hung upload fails the cycle into the existing degrade
404
- * handling with a named error, instead of stranding `refreshing` until the
405
- * 30-min watchdog. Restores are NOT on this budget any more: a download is
406
- * judged by the bytes arriving in its target (restoreWithProgress) —
407
- * a fixed budget abandoned a 481 s restore that then completed.
408
- * Same class as the other network budgets (observed live transfers run
409
- * seconds, recorded in `lastRestore.ms`). */
410
- const R2_TRANSFER_TIMEOUT_MS = 5 * 60_000;
411
382
  /** The mirror-mutex lease for a section that names no step budget of its own
412
383
  * (attach's clone section, a sweep's eviction, the wake and reclaim fetches):
413
384
  * a holder of the current incarnation still holding past this has hung, the
414
385
  * same bound the watchdog puts on a mid-flight state. The engine steps pass
415
386
  * their exact budgets instead. */
416
387
  const MIRROR_LEASE_DEFAULT_MS = STALE_MIDFLIGHT_MS;
417
- /** What the dependency install step runs around the install itself, each
418
- * bounded: the scratch clone, the seed and its cache swap, the commit (one
419
- * network budget each) and the harden (the default exec budget). The step's
420
- * lease is the install budget plus this. */
421
- const DEPS_STEP_OVERHEAD_MS = 4 * GIT_NETWORK_TIMEOUT_MS + DEFAULT_EXEC_TIMEOUT_MS;
422
388
 
423
389
  const sleep = (ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms));
424
390
 
@@ -507,7 +473,6 @@ const SWEEP_DRIFT_SLACK_S = 5 * 60;
507
473
  * the container can actually sleep (SLEEP_AFTER); the next attach refreshes
508
474
  * first if the mirror is stale (refresh-on-attach). */
509
475
  const IDLE_AFTER_S = 60 * 60;
510
- const IDLE_REFRESH_INTERVAL_S = 6 * 60 * 60;
511
476
  /** LRU eviction floor: an over-cap onboard with `evictColdest:true` may
512
477
  * offboard the coldest eligible warm resident, but never one whose last
513
478
  * activity (attach or provisioning) is younger than this — a repo used
@@ -535,7 +500,7 @@ const DEGRADED_STREAK_KEY = "resident:degradedStreak";
535
500
  /** Plus a cycle whose step was killed from OUTSIDE by a deploy
536
501
  * (`refresh-interrupted: …`, classified by `classifyRefreshFailure`): equally
537
502
  * not evidence about the repository, equally never counted. */
538
- const NON_EVIDENCE_REASON = /^(?:alarm-missed|stale-mid-flight|refresh-interrupted):/;
503
+ const NON_EVIDENCE_REASON = /^(?:alarm-missed|stale-mid-flight|refresh-interrupted|restore-interrupted):/;
539
504
  /** Consecutive cycles that ended `refresh-interrupted`: feeds the
540
505
  * short-re-arm cap in `nextRefreshDelayS`; cleared by any other outcome. */
541
506
  const INTERRUPTED_STREAK_KEY = "resident:interruptedStreak";
@@ -563,13 +528,6 @@ const LIFECYCLE_KEY = "resident:lifecycle";
563
528
  * this resident, with the step it last reported and the cycle lease it holds,
564
529
  * and the last bucket the cron skipped (a live cycle, a duplicate id). */
565
530
  const REFRESH_INSTANCE_KEY = "resident:refreshInstance";
566
- /** The refresh instance's step budgets: each `step.do` timeout is the DO
567
- * method's own budget, capped at the engine's 30-minute step ceiling
568
- * (`stepTimeoutMs`), so a step timeout and a command timeout agree. */
569
- const REFRESH_FETCH_STEP_BUDGET_MS = RESTORE_MAX_MS + GIT_NETWORK_TIMEOUT_MS; // a wake's restore, then the fetch
570
- const REFRESH_INSTALL_STEP_BUDGET_MS = REFRESH_INSTALL_TIMEOUT_MS + DEPS_STEP_OVERHEAD_MS; // the install's own lease
571
- const REFRESH_BUILD_STEP_BUDGET_MS = GIT_NETWORK_TIMEOUT_MS + REFRESH_BUILD_TIMEOUT_MS; // the build's mutex lease
572
- const REFRESH_SNAPSHOT_STEP_BUDGET_MS = R2_TRANSFER_TIMEOUT_MS + GIT_NETWORK_TIMEOUT_MS; // the archives, then the reclaim pass and the disk sample
573
531
  const DISK_MEASURE_CALLBACK = "onDiskMeasure";
574
532
  const DISK_MEASURE_DELAY_S = 1;
575
533
  /** A `du` over a multi-GB checkout plus every live tree is seconds warm, tens
@@ -655,8 +613,6 @@ export function validateEnvNames(vars: Record<string, string>): void {
655
613
  }
656
614
  }
657
615
 
658
- const errMsg = (err: unknown): string => (err instanceof Error ? err.message : String(err));
659
-
660
616
  /** The resident runtime (the Sandbox SDK's control session to the container)
661
617
  * was replaced while a command was in flight — in practice a `wrangler deploy`
662
618
  * swapping this DO's isolate mid-run (which otherwise surfaces as a fake
@@ -1130,7 +1086,7 @@ interface RefreshInstanceRow {
1130
1086
  /** What every instance step answers besides its own facts: the resident's
1131
1087
  * wall clock at the step's start and the commands it ran, so the instance
1132
1088
  * can graft them under its root the way the bot grafts an attach's. */
1133
- interface InstanceStepTrace {
1089
+ export interface InstanceStepTrace {
1134
1090
  startedAt: number;
1135
1091
  trace: ResidentStep[];
1136
1092
  }
@@ -1153,12 +1109,6 @@ interface RefreshFetchFacts {
1153
1109
  install: boolean;
1154
1110
  mintError: string | null;
1155
1111
  }
1156
- /** What the cron did about one resident's refresh instance this pass. */
1157
- interface RefreshInstanceAction {
1158
- id: string;
1159
- action: "created" | "duplicate" | "skipped" | "failed";
1160
- why: string;
1161
- }
1162
1112
 
1163
1113
  // ---------------------------------------------------------------------------
1164
1114
  // Registry DO (singleton): onboarded set + config, atomic cap enforcement
@@ -2251,9 +2201,13 @@ export class ResidentDO extends Sandbox<Env> {
2251
2201
  * their bytes, against the caller's one deadline), verify the restored
2252
2202
  * mirror against the stamp and hand the checkout to the build user. Runs
2253
2203
  * inside the hydration lease its caller holds — every mirror-mutex taker
2254
- * hydrates first, so nothing else touches these trees meanwhile. Failures
2255
- * leave the resident `down` with the reason and the container stopped, as
2256
- * the wake path always did. */
2204
+ * hydrates first, so nothing else touches these trees meanwhile. A stalled
2205
+ * or capped restore leaves the resident `down` with the reason and the
2206
+ * container stopped, as the wake path always did; a restore the runtime
2207
+ * replacement interrupts (a deploy rolled the container under it) is
2208
+ * `restore-interrupted`, degraded and rethrown for the cycle to re-arm
2209
+ * short — nothing is streaming into a disk that no longer exists
2210
+ * (restoreFailureDisposition). */
2257
2211
  async restoreCheckout(snap: SnapshotRecord, deadlineMs: number): Promise<{ done: boolean }> {
2258
2212
  const plan = planRestore({ sha: snap.sha, readyStamp: await this.readyStamp() });
2259
2213
  if (plan.action === "done") return { done: true };
@@ -2301,6 +2255,27 @@ export class ResidentDO extends Sandbox<Env> {
2301
2255
  deadlineMs,
2302
2256
  );
2303
2257
  } catch (err) {
2258
+ // The typed and cause-chain check first: the SDK's replacement errors
2259
+ // (a stale process handle, an inactive runtime identity, an interrupted
2260
+ // operation) carry the wording one cause down or not at all.
2261
+ const disposition = restoreFailureDisposition(errMsg(err), { runtimeReplaced: isRuntimeReplacement(err) });
2262
+ if (disposition.action === "interrupted") {
2263
+ // The runtime was replaced under the restore (a resident Worker deploy
2264
+ // rolled the container): the disk the stream wrote to is gone with the
2265
+ // container, so nothing can land on a rebuild and there is nothing to
2266
+ // stop. Not evidence about the repo — the resident is `degraded` with
2267
+ // the restore named, never `down`, and the error goes back to the
2268
+ // cycle, whose classifier reads the same wording as an interruption
2269
+ // and re-arms short; the next wake restores again onto the new
2270
+ // container. Before this branch every such restore ended `down`, and
2271
+ // only a rebuild (the watchdog's, after three passes) brought the
2272
+ // resident back.
2273
+ this.swapIncarnation(); // the container this incarnation's memos described is gone
2274
+ console.log(`restore: interrupted by a runtime replacement — ${disposition.reason.slice(0, 400)}`);
2275
+ await this.recordRefreshError(disposition.reason);
2276
+ await this.setResidentState("degraded", disposition.reason);
2277
+ throw err;
2278
+ }
2304
2279
  // A stalled or capped restore is STILL STREAMING (the SDK call cannot be
2305
2280
  // cancelled); `pendingRestores` keeps the next hydrate off its directory,
2306
2281
  // but a `down` resident's only exit is a REBUILD, and provisioning owns
@@ -2312,9 +2287,7 @@ export class ResidentDO extends Sandbox<Env> {
2312
2287
  // with it, and the rebuild starts on an empty one.
2313
2288
  this.swapIncarnation(); // deliberate incarnation swap
2314
2289
  await this.stop().catch((stopErr) => console.log(`restore: stop failed: ${errMsg(stopErr)}`));
2315
- throw await this.goDown(
2316
- `r2-restore-failed: ${errMsg(err)} — container stopped so the transfer cannot land on a rebuild`,
2317
- );
2290
+ throw await this.goDown(disposition.reason);
2318
2291
  }
2319
2292
  await this.ensureGitSetup();
2320
2293
 
@@ -2550,9 +2523,11 @@ export class ResidentDO extends Sandbox<Env> {
2550
2523
  /** Ensure the container disk holds the stamped snapshot state. `restoring`
2551
2524
  * is persisted BEFORE any restore work — DO storage would
2552
2525
  * otherwise still say warm while the R2 restore runs. Refuses mismatched
2553
- * stamps → down(snapshot-stamp-mismatch); restore failures →
2554
- * down(r2-restore-failed). Throws ResidentDownError after those
2555
- * transitions. Called by the refresh alarm (and the attach path). */
2526
+ * stamps → down(snapshot-stamp-mismatch); a stalled or capped restore →
2527
+ * down(r2-restore-failed); a restore the runtime replacement interrupts →
2528
+ * degraded(restore-interrupted), rethrown so the cycle re-arms short.
2529
+ * Throws ResidentDownError after the down transitions. Called by the
2530
+ * refresh alarm (and the attach path). */
2556
2531
  async ensureHydrated(): Promise<void> {
2557
2532
  // Fresh positive verdict for this incarnation → nothing to probe. See the
2558
2533
  // per-incarnation memo block for why this is safe; the refresh alarm's
@@ -3152,7 +3127,7 @@ export class ResidentDO extends Sandbox<Env> {
3152
3127
 
3153
3128
  // -- the refresh cycle as a Workflow instance (item 7) --------------------------
3154
3129
  //
3155
- // `ResidentRefresh` (the Workflow entrypoint, below the DO) calls these four
3130
+ // `ResidentRefresh` (the Workflow entrypoint, refresh.ts) calls these four
3156
3131
  // methods, one per step, through the DO stub. Each runs the same phase the
3157
3132
  // alarm runs, over the same rows, so a step the engine retries re-enters
3158
3133
  // the same idempotent read-then-act method (item 22) and finds the work
@@ -6652,14 +6627,6 @@ const ROUTES: Record<string, { scope: Scope; method: string }> = {
6652
6627
  "/op": { scope: "operator", method: "POST" },
6653
6628
  };
6654
6629
 
6655
- function registryStub(env: Env) {
6656
- return env.REGISTRY.get(env.REGISTRY.idFromName("registry"));
6657
- }
6658
-
6659
- function residentStub(env: Env, resource: string) {
6660
- return getSandbox(env.RESIDENT, resource, { sleepAfter: SLEEP_AFTER });
6661
- }
6662
-
6663
6630
  /** Per-resource R2 prefix for future resident cache objects; offboard deletes
6664
6631
  * everything beneath it. NOTE: SDK backup snapshots deliberately do NOT live
6665
6632
  * here — they land under backups/<uuid>/ and are deleted via the stored
@@ -7604,64 +7571,6 @@ async function handleDebug(env: Env, body: Record<string, unknown>): Promise<Res
7604
7571
  }
7605
7572
  }
7606
7573
 
7607
- /** The cron's instance-creation duty for one resident (item 7). Only a
7608
- * `workflow` row gets an instance, at most one per ten-minute bucket, never
7609
- * while a cycle is live (`shouldCreateRefreshInstance`); the id is
7610
- * deterministic per resident and bucket, so a second firing in one bucket
7611
- * meets the engine's duplicate-id refusal, which is the expected no-op. A
7612
- * skipped live cycle and a duplicate are recorded on the row for `/status`.
7613
- * An `alarm` resident answers null: nothing here touches it. */
7614
- /** Whether the engine knows an instance by this id, in any status. A missing
7615
- * id rejects on `get` or on `status`; either way the answer is false. */
7616
- async function refreshInstanceExists(env: Env, id: string): Promise<boolean> {
7617
- try {
7618
- await (await env.RESIDENT_REFRESH.get(id)).status();
7619
- return true;
7620
- } catch {
7621
- return false;
7622
- }
7623
- }
7624
-
7625
- async function createRefreshInstance(
7626
- env: Env,
7627
- stub: ReturnType<typeof residentStub>,
7628
- resource: string,
7629
- row: RefreshRow,
7630
- ): Promise<RefreshInstanceAction | null> {
7631
- if (row.lifecycle !== "workflow") return null;
7632
- const now = systemClock();
7633
- const decision = shouldCreateRefreshInstance(row, now, {
7634
- intervalS: REFRESH_INTERVAL_S,
7635
- idleIntervalS: IDLE_REFRESH_INTERVAL_S,
7636
- });
7637
- const slug = resource.slice("repo:".length);
7638
- const slash = slug.indexOf("/");
7639
- const id = refreshInstanceId(slug.slice(0, slash), slug.slice(slash + 1), now);
7640
- if (!decision.create) {
7641
- if (decision.why === "mid-cycle" || decision.why === "running")
7642
- await stub.recordRefreshSkipped(id, now, decision.why);
7643
- return { id, action: "skipped", why: decision.why };
7644
- }
7645
- try {
7646
- await env.RESIDENT_REFRESH.create({ id, params: { resource } });
7647
- } catch (err) {
7648
- const message = errMsg(err);
7649
- // The engine refuses an id that names an instance still inside its
7650
- // retention, and the refusal carries no code — so the id is asked, not
7651
- // the wording: an instance that answers for it exists, and the refusal
7652
- // was the duplicate it looks like. Any other failure stays a failure.
7653
- if (await refreshInstanceExists(env, id)) {
7654
- await stub.recordRefreshSkipped(id, now, "duplicate");
7655
- return { id, action: "duplicate", why: "duplicate" };
7656
- }
7657
- console.error(`resident-watchdog: creating refresh instance ${id} failed — ${message}`);
7658
- return { id, action: "failed", why: residentText(message) };
7659
- }
7660
- await stub.recordRefreshInstance(id, now);
7661
- console.log(`resident-watchdog: created refresh instance ${id}`);
7662
- return { id, action: "created", why: decision.why };
7663
- }
7664
-
7665
7574
  /** One watchdog pass over every registered resident. Shared by the cron
7666
7575
  * handler and the /debug run-watchdog op. Each check targets a different DO,
7667
7576
  * so they run concurrently; a failing one becomes its own {error} entry
@@ -7707,133 +7616,6 @@ async function runWatchdog(env: Env, parent?: TraceSpan): Promise<WatchdogSummar
7707
7616
  return { cap: (await registry.limits()).cap, count: residents.length, results };
7708
7617
  }
7709
7618
 
7710
- // ---------------------------------------------------------------------------
7711
- // The refresh cycle as a Workflow instance (docs/reference/specs/resident-repos.md item 7)
7712
- // ---------------------------------------------------------------------------
7713
-
7714
- /** What an instance answers when it ends: small facts for the engine's record. */
7715
- interface RefreshInstanceSummary {
7716
- instance: string;
7717
- /** `ok`, or the word a gate or a failure ended the cycle with. */
7718
- outcome: string;
7719
- step: "fetch" | "install" | "build" | "snapshot";
7720
- action?: RefreshPlan["action"];
7721
- sha?: string;
7722
- }
7723
-
7724
- /** One refresh cycle as one short Workflow instance: `fetch`, `install` (only
7725
- * when the plan moved the lockfile key), `build` (only when the branch
7726
- * moved), `snapshot` — each a `step.do` calling the resident's own step
7727
- * method through the DO stub, under the retry policy `REFRESH_STEP_RETRIES`
7728
- * (six attempts, thirty seconds apart, doubling: about 15.5 minutes, past
7729
- * the 3 to 10 minutes a resident Worker rollover takes to settle) and a
7730
- * timeout equal to the method's own budget (never above the engine's 30
7731
- * minutes). Inputs to a step are the event's `resource`, the instance id and
7732
- * previous steps' returns — refs, shas, a key, a path — never a payload and
7733
- * never a credential. A step killed from outside (the container replaced
7734
- * under it) throws and the engine retries it into the same idempotent
7735
- * method; a gate that ends the cycle (idle, a container restart) or a
7736
- * failure of the repository's own (recorded as `degraded`, the last snapshot
7737
- * still serving) ends the instance with that word, and the next cron firing
7738
- * creates the next one from the row's state. The instance runs one cycle
7739
- * and returns: it is created by the watchdog cron per resident and
7740
- * ten-minute bucket (`createRefreshInstance`), never a loop.
7741
- *
7742
- * The run is the cycle's root span, `resident.refresh` carrying the instance
7743
- * id (docs/reference/specs/tracing.md item 25), with every command a step ran
7744
- * grafted under it as a `resident.<step>` child — the same shape the alarm's
7745
- * root has. */
7746
- export class ResidentRefresh extends WorkflowEntrypoint<Env, RefreshInstanceParams> {
7747
- async run(
7748
- event: Readonly<WorkflowEvent<RefreshInstanceParams>>,
7749
- step: WorkflowStep,
7750
- ): Promise<RefreshInstanceSummary> {
7751
- const { resource } = event.payload;
7752
- const instance = event.instanceId;
7753
- const stub = residentStub(this.env, resource);
7754
- const root = startAdoptedRoot(tracer, "resident.refresh", {
7755
- sinks: traceSinks,
7756
- startedAt: event.timestamp.getTime(),
7757
- attrs: { instanceId: instance },
7758
- });
7759
- const graft = (answer: InstanceStepTrace) =>
7760
- graftResidentSteps(answer.trace, {
7761
- parent: root,
7762
- prefix: "resident",
7763
- baseAt: answer.startedAt,
7764
- clipAt: systemClock(),
7765
- });
7766
- const retries = REFRESH_STEP_RETRIES;
7767
- /** The word a step that did not finish ends the instance with, and its outcome for the root. */
7768
- const ended = (
7769
- at: RefreshInstanceSummary["step"],
7770
- answer: { status: "stopped"; why: string } | { status: "failed"; reason: string },
7771
- ): RefreshInstanceSummary => ({
7772
- instance,
7773
- outcome: answer.status === "stopped" ? answer.why : "failed",
7774
- step: at,
7775
- });
7776
- let summary: RefreshInstanceSummary | undefined;
7777
- try {
7778
- const fetched = await step.do("fetch", { retries, timeout: stepTimeoutMs(REFRESH_FETCH_STEP_BUDGET_MS) }, () =>
7779
- stub.refreshInstanceFetch({ resource, instance }),
7780
- );
7781
- graft(fetched);
7782
- if (fetched.status !== "done") return (summary = ended("fetch", fetched));
7783
- let depsEntry: string | null = null;
7784
- if (fetched.install) {
7785
- const installed = await step.do(
7786
- "install",
7787
- { retries, timeout: stepTimeoutMs(REFRESH_INSTALL_STEP_BUDGET_MS) },
7788
- () => stub.refreshInstanceInstall({ resource, instance, sha: fetched.sha, lockfileKey: fetched.lockfileKey }),
7789
- );
7790
- graft(installed);
7791
- if (installed.status !== "done") return (summary = ended("install", installed));
7792
- depsEntry = installed.entry;
7793
- }
7794
- if (fetched.action !== "unchanged") {
7795
- const built = await step.do("build", { retries, timeout: stepTimeoutMs(REFRESH_BUILD_STEP_BUDGET_MS) }, () =>
7796
- stub.refreshInstanceBuild({
7797
- resource,
7798
- instance,
7799
- sha: fetched.sha,
7800
- factsSha: fetched.factsSha,
7801
- lockfileKey: fetched.lockfileKey,
7802
- depsEntry,
7803
- }),
7804
- );
7805
- graft(built);
7806
- if (built.status !== "done") return (summary = ended("build", built));
7807
- }
7808
- const snapped = await step.do(
7809
- "snapshot",
7810
- { retries, timeout: stepTimeoutMs(REFRESH_SNAPSHOT_STEP_BUDGET_MS) },
7811
- () =>
7812
- stub.refreshInstanceSnapshot({
7813
- resource,
7814
- instance,
7815
- ref: fetched.ref,
7816
- sha: fetched.sha,
7817
- lockfileKey: fetched.lockfileKey,
7818
- action: fetched.action,
7819
- mintError: fetched.mintError,
7820
- }),
7821
- );
7822
- graft(snapped);
7823
- if (snapped.status !== "done") return (summary = ended("snapshot", snapped));
7824
- return (summary = { instance, outcome: "ok", step: "snapshot", action: fetched.action, sha: fetched.sha });
7825
- } catch (err) {
7826
- // A step out of retries: the engine records the failed instance by id;
7827
- // the row keeps its last state and the next cron firing starts the next cycle.
7828
- root.fail(err);
7829
- throw err;
7830
- } finally {
7831
- const outcome = summary?.outcome ?? "error";
7832
- root.end(outcome === "ok" || (summary !== undefined && outcome !== "failed") ? "ok" : "error", { outcome });
7833
- }
7834
- }
7835
- }
7836
-
7837
7619
  function json(data: unknown, status = 200): Response {
7838
7620
  // Item 62: every non-streamed body leaves through here; its `error`,
7839
7621
  // `reason` and `summary` strings are made safe at the exit.
@@ -111,7 +111,8 @@
111
111
  // account never share one — and BACKUP_BUCKET_NAME must say the same name.
112
112
  "r2_buckets": [{ "binding": "BACKUP_BUCKET", "bucket_name": "{{script}}-cache" }],
113
113
  // The refresh cycle as a Workflow instance (docs/reference/specs/resident-repos.md
114
- // item 7): `ResidentRefresh` in worker.ts runs one cycle — fetch, install,
114
+ // item 7): `ResidentRefresh` in refresh.ts, re-exported by worker.ts (the
115
+ // entry `class_name` resolves against), runs one cycle — fetch, install,
115
116
  // build, snapshot — as steps the engine retries and records, calling the
116
117
  // resident's own step methods. The watchdog cron below creates one instance
117
118
  // per resident whose row says `lifecycle: workflow` (the admin `/debug`