@coreplane/switchboard 1.201.0 → 1.202.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/bin/switchboard.js +26 -0
- package/dist/assets/deploy/cloudflare-resident/refresh.ts +238 -0
- package/dist/assets/deploy/cloudflare-resident/shared.ts +80 -0
- package/dist/assets/deploy/cloudflare-resident/tsconfig.json +1 -1
- package/dist/assets/deploy/cloudflare-resident/worker.ts +67 -285
- package/dist/assets/deploy/cloudflare-resident/wrangler.template.jsonc +2 -1
- package/dist/assets/package-lock.json +481 -4
- package/dist/assets/package.json +1 -1
- package/dist/assets/project.json +1 -1
- package/dist/assets/source.json +3 -3
- package/dist/assets/src/core/authz/policy.ts +3 -0
- package/dist/assets/src/execution/residentRefresh.ts +29 -0
- package/dist/assets/web/dist/.vite/manifest.json +43 -43
- package/dist/assets/web/dist/assets/{AppShell-CjDyTiip.js → AppShell-CJLPO_aI.js} +1 -1
- package/dist/assets/web/dist/assets/{CostsPage-CMAh3G74.js → CostsPage-C6ZvnAmH.js} +1 -1
- package/dist/assets/web/dist/assets/{NotFoundPage-CbEKm0tY.js → NotFoundPage-BdLcY60r.js} +1 -1
- package/dist/assets/web/dist/assets/{ResidentDetailPage-CcfmOAIP.js → ResidentDetailPage-hXWrr8T6.js} +1 -1
- package/dist/assets/web/dist/assets/{ResidentsIndexPage-9Z5jcteM.js → ResidentsIndexPage-YLu4JwxJ.js} +1 -1
- package/dist/assets/web/dist/assets/{RunRoutePage-Cdimh4e0.js → RunRoutePage-BDhHRJVO.js} +1 -1
- package/dist/assets/web/dist/assets/{RunsIndexPage-CvgDXrO4.js → RunsIndexPage-D7zZ5a0l.js} +1 -1
- package/dist/assets/web/dist/assets/{RunsTabs-rEjVQG6H.js → RunsTabs-BOUlSa2W.js} +1 -1
- package/dist/assets/web/dist/assets/{ScheduledPage-C5bG9qvd.js → ScheduledPage-BciOQcC-.js} +1 -1
- package/dist/assets/web/dist/assets/{StatusDot-D_IPK3WP.js → StatusDot-CTDLBv92.js} +1 -1
- package/dist/assets/web/dist/assets/{Tooltip-JlZp3OVC.js → Tooltip-DD2v9Gxx.js} +1 -1
- package/dist/assets/web/dist/assets/{favicon-DZDjc8ab.js → favicon-C4Q8cRsP.js} +1 -1
- package/dist/assets/web/dist/assets/{main-BKKzFAnK.js → main-DrSlUcMg.js} +2 -2
- package/dist/cli.js +629 -207
- package/package.json +3 -2
|
@@ -65,7 +65,6 @@
|
|
|
65
65
|
// (untrusted repo code) run unprivileged (worker1) and token-free
|
|
66
66
|
// (docs/decisions/0009-residents-second-credential-domain.md).
|
|
67
67
|
import {
|
|
68
|
-
getSandbox,
|
|
69
68
|
isDurableObjectCodeUpdateReset,
|
|
70
69
|
OperationInterruptedError,
|
|
71
70
|
ProcessWaitTimeoutError,
|
|
@@ -77,7 +76,7 @@ import {
|
|
|
77
76
|
import { AsyncLocalStorage } from "node:async_hooks";
|
|
78
77
|
import type { DirectoryBackup, SandboxCommand } from "@cloudflare/sandbox";
|
|
79
78
|
import { createExtensionProcessSandbox } from "@cloudflare/sandbox/extensions";
|
|
80
|
-
import { DurableObject
|
|
79
|
+
import { DurableObject } from "cloudflare:workers";
|
|
81
80
|
import { BASH_TIMEOUT_MAX_MS, BASH_TIMEOUT_MS, clampBashTimeout } from "../../src/execution/bashTimeout.js";
|
|
82
81
|
import { selectBindingsToPurge } from "../../src/execution/bindingPurge.js";
|
|
83
82
|
import { busyAfterKillReason, planForceDetach } from "../../src/execution/residentDetach.js";
|
|
@@ -130,6 +129,7 @@ import {
|
|
|
130
129
|
import {
|
|
131
130
|
checkoutUpdateCommand,
|
|
132
131
|
classifyRefreshFailure,
|
|
132
|
+
restoreFailureDisposition,
|
|
133
133
|
killStaleBuildProcessesCommand,
|
|
134
134
|
nextRefreshDelayS,
|
|
135
135
|
planRefresh,
|
|
@@ -147,12 +147,8 @@ import {
|
|
|
147
147
|
type RefreshOutcome,
|
|
148
148
|
} from "../../src/execution/residentRefresh.js";
|
|
149
149
|
import {
|
|
150
|
-
REFRESH_STEP_RETRIES,
|
|
151
150
|
lifecycleOf,
|
|
152
151
|
parseLifecycle,
|
|
153
|
-
refreshInstanceId,
|
|
154
|
-
shouldCreateRefreshInstance,
|
|
155
|
-
stepTimeoutMs,
|
|
156
152
|
type RefreshRow,
|
|
157
153
|
type ResidentLifecycle,
|
|
158
154
|
} from "../../src/execution/residentInstanceId.js";
|
|
@@ -221,8 +217,7 @@ import { graftResidentSteps } from "../../src/execution/residentTrace.js";
|
|
|
221
217
|
import type { SpanAttrs } from "../../src/core/trace/attrs.js";
|
|
222
218
|
import type { Span as TraceSpan } from "../../src/core/trace/types.js";
|
|
223
219
|
import { systemClock } from "../../src/core/trace/clock.js";
|
|
224
|
-
import {
|
|
225
|
-
import { startAdoptedRoot, workerLogSink } from "../../src/core/trace/workerTrace.js";
|
|
220
|
+
import { startAdoptedRoot } from "../../src/core/trace/workerTrace.js";
|
|
226
221
|
import { backupTransferMode } from "../../src/execution/residentBackupTransfer.js";
|
|
227
222
|
import {
|
|
228
223
|
extractRestoreScript,
|
|
@@ -255,6 +250,27 @@ import {
|
|
|
255
250
|
planDepsMaterialization,
|
|
256
251
|
} from "../../src/execution/residentDepsStore.js";
|
|
257
252
|
import { buildId, injectedBuildStamp } from "../../src/deploy/buildStamp.js";
|
|
253
|
+
import { createRefreshInstance, type RefreshInstanceParams } from "./refresh";
|
|
254
|
+
import {
|
|
255
|
+
DEFAULT_EXEC_TIMEOUT_MS,
|
|
256
|
+
DEPS_STEP_OVERHEAD_MS,
|
|
257
|
+
errMsg,
|
|
258
|
+
GIT_NETWORK_TIMEOUT_MS,
|
|
259
|
+
IDLE_REFRESH_INTERVAL_S,
|
|
260
|
+
R2_TRANSFER_TIMEOUT_MS,
|
|
261
|
+
REFRESH_BUILD_TIMEOUT_MS,
|
|
262
|
+
REFRESH_INSTALL_TIMEOUT_MS,
|
|
263
|
+
REFRESH_INTERVAL_S,
|
|
264
|
+
registryStub,
|
|
265
|
+
residentStub,
|
|
266
|
+
tracer,
|
|
267
|
+
traceSinks,
|
|
268
|
+
} from "./shared";
|
|
269
|
+
|
|
270
|
+
/** The refresh cycle's Workflow entrypoint is declared in refresh.ts; the
|
|
271
|
+
* Workflows binding resolves its `class_name` against this module
|
|
272
|
+
* (wrangler.template.jsonc), so the entry exports it under that name. */
|
|
273
|
+
export { ResidentRefresh } from "./refresh";
|
|
258
274
|
|
|
259
275
|
/** The commit this bundle was built from, injected by the deploy
|
|
260
276
|
* (`deploy/bin/build-stamp.mjs`; `unknown` when nobody stamped it). Answered
|
|
@@ -270,9 +286,8 @@ const BUILD_ID = buildId(BUILD);
|
|
|
270
286
|
// (/attach, /exec, /op) are rooted inside the DO where the work is, their
|
|
271
287
|
// collector's steps as `resident.<step>` children; every other authenticated
|
|
272
288
|
// route is a `resident.fetch` root at the edge. Each joins the bot's trace when
|
|
273
|
-
// the request carried one.
|
|
274
|
-
|
|
275
|
-
const traceSinks = [workerLogSink((line) => console.log(line))];
|
|
289
|
+
// the request carried one. The tracer and its log sink are shared.ts's: the
|
|
290
|
+
// refresh instance's root (refresh.ts) starts from the same pair.
|
|
276
291
|
const STREAMED_ROUTES: ReadonlySet<string> = new Set(["/attach", "/exec", "/op"]);
|
|
277
292
|
|
|
278
293
|
/** One request as the resident's own root: started at its t0, joining the
|
|
@@ -296,20 +311,14 @@ function refusalOutcome(err: ThreadErr): string {
|
|
|
296
311
|
return err.needs ? `needs_${err.needs}` : "error";
|
|
297
312
|
}
|
|
298
313
|
|
|
299
|
-
|
|
300
|
-
* other input is read from the resident's rows at each step, never carried. */
|
|
301
|
-
interface RefreshInstanceParams {
|
|
302
|
-
resource: string;
|
|
303
|
-
}
|
|
304
|
-
|
|
305
|
-
interface Env {
|
|
314
|
+
export interface Env {
|
|
306
315
|
RESIDENT: DurableObjectNamespace<ResidentDO>;
|
|
307
316
|
REGISTRY: DurableObjectNamespace<ResidentRegistryDO>;
|
|
308
317
|
BACKUP_BUCKET: R2Bucket;
|
|
309
318
|
/** The refresh cycle as a Workflow instance (docs/reference/specs/resident-repos.md
|
|
310
|
-
* item 7): `ResidentRefresh`
|
|
311
|
-
* resident whose row says `lifecycle: workflow`; an
|
|
312
|
-
* default) never has one. */
|
|
319
|
+
* item 7): `ResidentRefresh` in refresh.ts, re-exported above. The watchdog
|
|
320
|
+
* cron creates one per resident whose row says `lifecycle: workflow`; an
|
|
321
|
+
* `alarm` resident (the default) never has one. */
|
|
313
322
|
RESIDENT_REFRESH: Workflow<RefreshInstanceParams>;
|
|
314
323
|
// Presigned snapshot transfers (docs/reference/specs/resident-repos.md item 61): with all
|
|
315
324
|
// four present the container moves archive bytes itself over presigned R2
|
|
@@ -354,19 +363,6 @@ interface Env {
|
|
|
354
363
|
// `evictColdest:true` makes room (docs/reference/specs/resident-repos.md item 46).
|
|
355
364
|
const RESIDENT_CAP = 6;
|
|
356
365
|
|
|
357
|
-
/** Container sleep window, passed to every getSandbox() for ResidentDO.
|
|
358
|
-
* Invariant: REFRESH_INTERVAL_S and the watchdog cron (wrangler.jsonc,
|
|
359
|
-
* every 10 minutes) MUST both stay SHORTER than this window, so a healthy
|
|
360
|
-
* resident is re-warmed before the platform can sleep it. Bump together. */
|
|
361
|
-
const SLEEP_AFTER = "20m";
|
|
362
|
-
|
|
363
|
-
/** Refresh alarm cadence (seconds). Each resident DO self-reschedules this
|
|
364
|
-
* alarm (per-resident alarms own freshness; the sparse cron is only the
|
|
365
|
-
* watchdog); it doubles as the keep-warm heartbeat, so it must stay below
|
|
366
|
-
* SLEEP_AFTER. Matches the watchdog cron so a killed chain is re-armed
|
|
367
|
-
* within one refresh interval. */
|
|
368
|
-
const REFRESH_INTERVAL_S = 600;
|
|
369
|
-
|
|
370
366
|
/** R2 lifetime of snapshot objects. We delete replaced/offboarded snapshots
|
|
371
367
|
* explicitly (see deleteBackupObjects); the TTL is a leak backstop, and it
|
|
372
368
|
* must be long — a quiet repo's current snapshot may go unreplaced for
|
|
@@ -379,46 +375,16 @@ const DEFAULT_PROVISIONING_TIMEOUT_MS = 5 * 60_000;
|
|
|
379
375
|
const MIN_PROVISIONING_TIMEOUT_MS = 10_000;
|
|
380
376
|
const MAX_PROVISIONING_TIMEOUT_MS = 30 * 60_000;
|
|
381
377
|
|
|
382
|
-
/** Exec budgets. The DO alarm handler has a ~15-minute platform wall clock;
|
|
383
|
-
* every schedule callback's step budgets are chosen to fit under it. */
|
|
384
|
-
const DEFAULT_EXEC_TIMEOUT_MS = 60_000;
|
|
385
|
-
const GIT_NETWORK_TIMEOUT_MS = 5 * 60_000;
|
|
386
|
-
const REFRESH_BUILD_TIMEOUT_MS = 5 * 60_000;
|
|
387
|
-
/** The refresh install budget. Twice the build's: a full `npm install` of the
|
|
388
|
-
* switchboard lockfile takes ~4 min on the resident's 1 vCPU when nothing
|
|
389
|
-
* else runs, and thread runs (tests, a review's greps) share that vCPU —
|
|
390
|
-
* under load it crosses 5 min several cycles in a row while the default
|
|
391
|
-
* branch keeps moving. A timed-out install is worse than a slow one: the cycle's whole
|
|
392
|
-
* budget is spent and the checkout is left without deps, so the next cycle
|
|
393
|
-
* starts the same install over. The cycle runs in the background (runs keep
|
|
394
|
-
* attaching to the last snapshot); the cost of a longer budget is a longer
|
|
395
|
-
* mirror-lock window, bounded well inside STALE_MIDFLIGHT_MS. */
|
|
396
|
-
const REFRESH_INSTALL_TIMEOUT_MS = 10 * 60_000;
|
|
397
378
|
/** After `output()`'s wait gives up on a process the supervisor should have
|
|
398
379
|
* killed at `timeout`, how long to wait for the exit status of OUR kill
|
|
399
380
|
* before reporting the step without one. */
|
|
400
381
|
const KILL_EXIT_WAIT_MS = 10_000;
|
|
401
|
-
/** Budget per R2 SNAPSHOT upload. The SDK's createBackup
|
|
402
|
-
* accepts no timeout or AbortSignal, so each call is raced against this
|
|
403
|
-
* (withTimeout): a hung upload fails the cycle into the existing degrade
|
|
404
|
-
* handling with a named error, instead of stranding `refreshing` until the
|
|
405
|
-
* 30-min watchdog. Restores are NOT on this budget any more: a download is
|
|
406
|
-
* judged by the bytes arriving in its target (restoreWithProgress) —
|
|
407
|
-
* a fixed budget abandoned a 481 s restore that then completed.
|
|
408
|
-
* Same class as the other network budgets (observed live transfers run
|
|
409
|
-
* seconds, recorded in `lastRestore.ms`). */
|
|
410
|
-
const R2_TRANSFER_TIMEOUT_MS = 5 * 60_000;
|
|
411
382
|
/** The mirror-mutex lease for a section that names no step budget of its own
|
|
412
383
|
* (attach's clone section, a sweep's eviction, the wake and reclaim fetches):
|
|
413
384
|
* a holder of the current incarnation still holding past this has hung, the
|
|
414
385
|
* same bound the watchdog puts on a mid-flight state. The engine steps pass
|
|
415
386
|
* their exact budgets instead. */
|
|
416
387
|
const MIRROR_LEASE_DEFAULT_MS = STALE_MIDFLIGHT_MS;
|
|
417
|
-
/** What the dependency install step runs around the install itself, each
|
|
418
|
-
* bounded: the scratch clone, the seed and its cache swap, the commit (one
|
|
419
|
-
* network budget each) and the harden (the default exec budget). The step's
|
|
420
|
-
* lease is the install budget plus this. */
|
|
421
|
-
const DEPS_STEP_OVERHEAD_MS = 4 * GIT_NETWORK_TIMEOUT_MS + DEFAULT_EXEC_TIMEOUT_MS;
|
|
422
388
|
|
|
423
389
|
const sleep = (ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms));
|
|
424
390
|
|
|
@@ -507,7 +473,6 @@ const SWEEP_DRIFT_SLACK_S = 5 * 60;
|
|
|
507
473
|
* the container can actually sleep (SLEEP_AFTER); the next attach refreshes
|
|
508
474
|
* first if the mirror is stale (refresh-on-attach). */
|
|
509
475
|
const IDLE_AFTER_S = 60 * 60;
|
|
510
|
-
const IDLE_REFRESH_INTERVAL_S = 6 * 60 * 60;
|
|
511
476
|
/** LRU eviction floor: an over-cap onboard with `evictColdest:true` may
|
|
512
477
|
* offboard the coldest eligible warm resident, but never one whose last
|
|
513
478
|
* activity (attach or provisioning) is younger than this — a repo used
|
|
@@ -535,7 +500,7 @@ const DEGRADED_STREAK_KEY = "resident:degradedStreak";
|
|
|
535
500
|
/** Plus a cycle whose step was killed from OUTSIDE by a deploy
|
|
536
501
|
* (`refresh-interrupted: …`, classified by `classifyRefreshFailure`): equally
|
|
537
502
|
* not evidence about the repository, equally never counted. */
|
|
538
|
-
const NON_EVIDENCE_REASON = /^(?:alarm-missed|stale-mid-flight|refresh-interrupted):/;
|
|
503
|
+
const NON_EVIDENCE_REASON = /^(?:alarm-missed|stale-mid-flight|refresh-interrupted|restore-interrupted):/;
|
|
539
504
|
/** Consecutive cycles that ended `refresh-interrupted`: feeds the
|
|
540
505
|
* short-re-arm cap in `nextRefreshDelayS`; cleared by any other outcome. */
|
|
541
506
|
const INTERRUPTED_STREAK_KEY = "resident:interruptedStreak";
|
|
@@ -563,13 +528,6 @@ const LIFECYCLE_KEY = "resident:lifecycle";
|
|
|
563
528
|
* this resident, with the step it last reported and the cycle lease it holds,
|
|
564
529
|
* and the last bucket the cron skipped (a live cycle, a duplicate id). */
|
|
565
530
|
const REFRESH_INSTANCE_KEY = "resident:refreshInstance";
|
|
566
|
-
/** The refresh instance's step budgets: each `step.do` timeout is the DO
|
|
567
|
-
* method's own budget, capped at the engine's 30-minute step ceiling
|
|
568
|
-
* (`stepTimeoutMs`), so a step timeout and a command timeout agree. */
|
|
569
|
-
const REFRESH_FETCH_STEP_BUDGET_MS = RESTORE_MAX_MS + GIT_NETWORK_TIMEOUT_MS; // a wake's restore, then the fetch
|
|
570
|
-
const REFRESH_INSTALL_STEP_BUDGET_MS = REFRESH_INSTALL_TIMEOUT_MS + DEPS_STEP_OVERHEAD_MS; // the install's own lease
|
|
571
|
-
const REFRESH_BUILD_STEP_BUDGET_MS = GIT_NETWORK_TIMEOUT_MS + REFRESH_BUILD_TIMEOUT_MS; // the build's mutex lease
|
|
572
|
-
const REFRESH_SNAPSHOT_STEP_BUDGET_MS = R2_TRANSFER_TIMEOUT_MS + GIT_NETWORK_TIMEOUT_MS; // the archives, then the reclaim pass and the disk sample
|
|
573
531
|
const DISK_MEASURE_CALLBACK = "onDiskMeasure";
|
|
574
532
|
const DISK_MEASURE_DELAY_S = 1;
|
|
575
533
|
/** A `du` over a multi-GB checkout plus every live tree is seconds warm, tens
|
|
@@ -655,8 +613,6 @@ export function validateEnvNames(vars: Record<string, string>): void {
|
|
|
655
613
|
}
|
|
656
614
|
}
|
|
657
615
|
|
|
658
|
-
const errMsg = (err: unknown): string => (err instanceof Error ? err.message : String(err));
|
|
659
|
-
|
|
660
616
|
/** The resident runtime (the Sandbox SDK's control session to the container)
|
|
661
617
|
* was replaced while a command was in flight — in practice a `wrangler deploy`
|
|
662
618
|
* swapping this DO's isolate mid-run (which otherwise surfaces as a fake
|
|
@@ -1130,7 +1086,7 @@ interface RefreshInstanceRow {
|
|
|
1130
1086
|
/** What every instance step answers besides its own facts: the resident's
|
|
1131
1087
|
* wall clock at the step's start and the commands it ran, so the instance
|
|
1132
1088
|
* can graft them under its root the way the bot grafts an attach's. */
|
|
1133
|
-
interface InstanceStepTrace {
|
|
1089
|
+
export interface InstanceStepTrace {
|
|
1134
1090
|
startedAt: number;
|
|
1135
1091
|
trace: ResidentStep[];
|
|
1136
1092
|
}
|
|
@@ -1153,12 +1109,6 @@ interface RefreshFetchFacts {
|
|
|
1153
1109
|
install: boolean;
|
|
1154
1110
|
mintError: string | null;
|
|
1155
1111
|
}
|
|
1156
|
-
/** What the cron did about one resident's refresh instance this pass. */
|
|
1157
|
-
interface RefreshInstanceAction {
|
|
1158
|
-
id: string;
|
|
1159
|
-
action: "created" | "duplicate" | "skipped" | "failed";
|
|
1160
|
-
why: string;
|
|
1161
|
-
}
|
|
1162
1112
|
|
|
1163
1113
|
// ---------------------------------------------------------------------------
|
|
1164
1114
|
// Registry DO (singleton): onboarded set + config, atomic cap enforcement
|
|
@@ -2251,9 +2201,13 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2251
2201
|
* their bytes, against the caller's one deadline), verify the restored
|
|
2252
2202
|
* mirror against the stamp and hand the checkout to the build user. Runs
|
|
2253
2203
|
* inside the hydration lease its caller holds — every mirror-mutex taker
|
|
2254
|
-
* hydrates first, so nothing else touches these trees meanwhile.
|
|
2255
|
-
*
|
|
2256
|
-
* the wake path always did
|
|
2204
|
+
* hydrates first, so nothing else touches these trees meanwhile. A stalled
|
|
2205
|
+
* or capped restore leaves the resident `down` with the reason and the
|
|
2206
|
+
* container stopped, as the wake path always did; a restore the runtime
|
|
2207
|
+
* replacement interrupts (a deploy rolled the container under it) is
|
|
2208
|
+
* `restore-interrupted`, degraded and rethrown for the cycle to re-arm
|
|
2209
|
+
* short — nothing is streaming into a disk that no longer exists
|
|
2210
|
+
* (restoreFailureDisposition). */
|
|
2257
2211
|
async restoreCheckout(snap: SnapshotRecord, deadlineMs: number): Promise<{ done: boolean }> {
|
|
2258
2212
|
const plan = planRestore({ sha: snap.sha, readyStamp: await this.readyStamp() });
|
|
2259
2213
|
if (plan.action === "done") return { done: true };
|
|
@@ -2301,6 +2255,27 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2301
2255
|
deadlineMs,
|
|
2302
2256
|
);
|
|
2303
2257
|
} catch (err) {
|
|
2258
|
+
// The typed and cause-chain check first: the SDK's replacement errors
|
|
2259
|
+
// (a stale process handle, an inactive runtime identity, an interrupted
|
|
2260
|
+
// operation) carry the wording one cause down or not at all.
|
|
2261
|
+
const disposition = restoreFailureDisposition(errMsg(err), { runtimeReplaced: isRuntimeReplacement(err) });
|
|
2262
|
+
if (disposition.action === "interrupted") {
|
|
2263
|
+
// The runtime was replaced under the restore (a resident Worker deploy
|
|
2264
|
+
// rolled the container): the disk the stream wrote to is gone with the
|
|
2265
|
+
// container, so nothing can land on a rebuild and there is nothing to
|
|
2266
|
+
// stop. Not evidence about the repo — the resident is `degraded` with
|
|
2267
|
+
// the restore named, never `down`, and the error goes back to the
|
|
2268
|
+
// cycle, whose classifier reads the same wording as an interruption
|
|
2269
|
+
// and re-arms short; the next wake restores again onto the new
|
|
2270
|
+
// container. Before this branch every such restore ended `down`, and
|
|
2271
|
+
// only a rebuild (the watchdog's, after three passes) brought the
|
|
2272
|
+
// resident back.
|
|
2273
|
+
this.swapIncarnation(); // the container this incarnation's memos described is gone
|
|
2274
|
+
console.log(`restore: interrupted by a runtime replacement — ${disposition.reason.slice(0, 400)}`);
|
|
2275
|
+
await this.recordRefreshError(disposition.reason);
|
|
2276
|
+
await this.setResidentState("degraded", disposition.reason);
|
|
2277
|
+
throw err;
|
|
2278
|
+
}
|
|
2304
2279
|
// A stalled or capped restore is STILL STREAMING (the SDK call cannot be
|
|
2305
2280
|
// cancelled); `pendingRestores` keeps the next hydrate off its directory,
|
|
2306
2281
|
// but a `down` resident's only exit is a REBUILD, and provisioning owns
|
|
@@ -2312,9 +2287,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2312
2287
|
// with it, and the rebuild starts on an empty one.
|
|
2313
2288
|
this.swapIncarnation(); // deliberate incarnation swap
|
|
2314
2289
|
await this.stop().catch((stopErr) => console.log(`restore: stop failed: ${errMsg(stopErr)}`));
|
|
2315
|
-
throw await this.goDown(
|
|
2316
|
-
`r2-restore-failed: ${errMsg(err)} — container stopped so the transfer cannot land on a rebuild`,
|
|
2317
|
-
);
|
|
2290
|
+
throw await this.goDown(disposition.reason);
|
|
2318
2291
|
}
|
|
2319
2292
|
await this.ensureGitSetup();
|
|
2320
2293
|
|
|
@@ -2550,9 +2523,11 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2550
2523
|
/** Ensure the container disk holds the stamped snapshot state. `restoring`
|
|
2551
2524
|
* is persisted BEFORE any restore work — DO storage would
|
|
2552
2525
|
* otherwise still say warm while the R2 restore runs. Refuses mismatched
|
|
2553
|
-
* stamps → down(snapshot-stamp-mismatch); restore
|
|
2554
|
-
* down(r2-restore-failed)
|
|
2555
|
-
*
|
|
2526
|
+
* stamps → down(snapshot-stamp-mismatch); a stalled or capped restore →
|
|
2527
|
+
* down(r2-restore-failed); a restore the runtime replacement interrupts →
|
|
2528
|
+
* degraded(restore-interrupted), rethrown so the cycle re-arms short.
|
|
2529
|
+
* Throws ResidentDownError after the down transitions. Called by the
|
|
2530
|
+
* refresh alarm (and the attach path). */
|
|
2556
2531
|
async ensureHydrated(): Promise<void> {
|
|
2557
2532
|
// Fresh positive verdict for this incarnation → nothing to probe. See the
|
|
2558
2533
|
// per-incarnation memo block for why this is safe; the refresh alarm's
|
|
@@ -3152,7 +3127,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3152
3127
|
|
|
3153
3128
|
// -- the refresh cycle as a Workflow instance (item 7) --------------------------
|
|
3154
3129
|
//
|
|
3155
|
-
// `ResidentRefresh` (the Workflow entrypoint,
|
|
3130
|
+
// `ResidentRefresh` (the Workflow entrypoint, refresh.ts) calls these four
|
|
3156
3131
|
// methods, one per step, through the DO stub. Each runs the same phase the
|
|
3157
3132
|
// alarm runs, over the same rows, so a step the engine retries re-enters
|
|
3158
3133
|
// the same idempotent read-then-act method (item 22) and finds the work
|
|
@@ -6652,14 +6627,6 @@ const ROUTES: Record<string, { scope: Scope; method: string }> = {
|
|
|
6652
6627
|
"/op": { scope: "operator", method: "POST" },
|
|
6653
6628
|
};
|
|
6654
6629
|
|
|
6655
|
-
function registryStub(env: Env) {
|
|
6656
|
-
return env.REGISTRY.get(env.REGISTRY.idFromName("registry"));
|
|
6657
|
-
}
|
|
6658
|
-
|
|
6659
|
-
function residentStub(env: Env, resource: string) {
|
|
6660
|
-
return getSandbox(env.RESIDENT, resource, { sleepAfter: SLEEP_AFTER });
|
|
6661
|
-
}
|
|
6662
|
-
|
|
6663
6630
|
/** Per-resource R2 prefix for future resident cache objects; offboard deletes
|
|
6664
6631
|
* everything beneath it. NOTE: SDK backup snapshots deliberately do NOT live
|
|
6665
6632
|
* here — they land under backups/<uuid>/ and are deleted via the stored
|
|
@@ -7604,64 +7571,6 @@ async function handleDebug(env: Env, body: Record<string, unknown>): Promise<Res
|
|
|
7604
7571
|
}
|
|
7605
7572
|
}
|
|
7606
7573
|
|
|
7607
|
-
/** The cron's instance-creation duty for one resident (item 7). Only a
|
|
7608
|
-
* `workflow` row gets an instance, at most one per ten-minute bucket, never
|
|
7609
|
-
* while a cycle is live (`shouldCreateRefreshInstance`); the id is
|
|
7610
|
-
* deterministic per resident and bucket, so a second firing in one bucket
|
|
7611
|
-
* meets the engine's duplicate-id refusal, which is the expected no-op. A
|
|
7612
|
-
* skipped live cycle and a duplicate are recorded on the row for `/status`.
|
|
7613
|
-
* An `alarm` resident answers null: nothing here touches it. */
|
|
7614
|
-
/** Whether the engine knows an instance by this id, in any status. A missing
|
|
7615
|
-
* id rejects on `get` or on `status`; either way the answer is false. */
|
|
7616
|
-
async function refreshInstanceExists(env: Env, id: string): Promise<boolean> {
|
|
7617
|
-
try {
|
|
7618
|
-
await (await env.RESIDENT_REFRESH.get(id)).status();
|
|
7619
|
-
return true;
|
|
7620
|
-
} catch {
|
|
7621
|
-
return false;
|
|
7622
|
-
}
|
|
7623
|
-
}
|
|
7624
|
-
|
|
7625
|
-
async function createRefreshInstance(
|
|
7626
|
-
env: Env,
|
|
7627
|
-
stub: ReturnType<typeof residentStub>,
|
|
7628
|
-
resource: string,
|
|
7629
|
-
row: RefreshRow,
|
|
7630
|
-
): Promise<RefreshInstanceAction | null> {
|
|
7631
|
-
if (row.lifecycle !== "workflow") return null;
|
|
7632
|
-
const now = systemClock();
|
|
7633
|
-
const decision = shouldCreateRefreshInstance(row, now, {
|
|
7634
|
-
intervalS: REFRESH_INTERVAL_S,
|
|
7635
|
-
idleIntervalS: IDLE_REFRESH_INTERVAL_S,
|
|
7636
|
-
});
|
|
7637
|
-
const slug = resource.slice("repo:".length);
|
|
7638
|
-
const slash = slug.indexOf("/");
|
|
7639
|
-
const id = refreshInstanceId(slug.slice(0, slash), slug.slice(slash + 1), now);
|
|
7640
|
-
if (!decision.create) {
|
|
7641
|
-
if (decision.why === "mid-cycle" || decision.why === "running")
|
|
7642
|
-
await stub.recordRefreshSkipped(id, now, decision.why);
|
|
7643
|
-
return { id, action: "skipped", why: decision.why };
|
|
7644
|
-
}
|
|
7645
|
-
try {
|
|
7646
|
-
await env.RESIDENT_REFRESH.create({ id, params: { resource } });
|
|
7647
|
-
} catch (err) {
|
|
7648
|
-
const message = errMsg(err);
|
|
7649
|
-
// The engine refuses an id that names an instance still inside its
|
|
7650
|
-
// retention, and the refusal carries no code — so the id is asked, not
|
|
7651
|
-
// the wording: an instance that answers for it exists, and the refusal
|
|
7652
|
-
// was the duplicate it looks like. Any other failure stays a failure.
|
|
7653
|
-
if (await refreshInstanceExists(env, id)) {
|
|
7654
|
-
await stub.recordRefreshSkipped(id, now, "duplicate");
|
|
7655
|
-
return { id, action: "duplicate", why: "duplicate" };
|
|
7656
|
-
}
|
|
7657
|
-
console.error(`resident-watchdog: creating refresh instance ${id} failed — ${message}`);
|
|
7658
|
-
return { id, action: "failed", why: residentText(message) };
|
|
7659
|
-
}
|
|
7660
|
-
await stub.recordRefreshInstance(id, now);
|
|
7661
|
-
console.log(`resident-watchdog: created refresh instance ${id}`);
|
|
7662
|
-
return { id, action: "created", why: decision.why };
|
|
7663
|
-
}
|
|
7664
|
-
|
|
7665
7574
|
/** One watchdog pass over every registered resident. Shared by the cron
|
|
7666
7575
|
* handler and the /debug run-watchdog op. Each check targets a different DO,
|
|
7667
7576
|
* so they run concurrently; a failing one becomes its own {error} entry
|
|
@@ -7707,133 +7616,6 @@ async function runWatchdog(env: Env, parent?: TraceSpan): Promise<WatchdogSummar
|
|
|
7707
7616
|
return { cap: (await registry.limits()).cap, count: residents.length, results };
|
|
7708
7617
|
}
|
|
7709
7618
|
|
|
7710
|
-
// ---------------------------------------------------------------------------
|
|
7711
|
-
// The refresh cycle as a Workflow instance (docs/reference/specs/resident-repos.md item 7)
|
|
7712
|
-
// ---------------------------------------------------------------------------
|
|
7713
|
-
|
|
7714
|
-
/** What an instance answers when it ends: small facts for the engine's record. */
|
|
7715
|
-
interface RefreshInstanceSummary {
|
|
7716
|
-
instance: string;
|
|
7717
|
-
/** `ok`, or the word a gate or a failure ended the cycle with. */
|
|
7718
|
-
outcome: string;
|
|
7719
|
-
step: "fetch" | "install" | "build" | "snapshot";
|
|
7720
|
-
action?: RefreshPlan["action"];
|
|
7721
|
-
sha?: string;
|
|
7722
|
-
}
|
|
7723
|
-
|
|
7724
|
-
/** One refresh cycle as one short Workflow instance: `fetch`, `install` (only
|
|
7725
|
-
* when the plan moved the lockfile key), `build` (only when the branch
|
|
7726
|
-
* moved), `snapshot` — each a `step.do` calling the resident's own step
|
|
7727
|
-
* method through the DO stub, under the retry policy `REFRESH_STEP_RETRIES`
|
|
7728
|
-
* (six attempts, thirty seconds apart, doubling: about 15.5 minutes, past
|
|
7729
|
-
* the 3 to 10 minutes a resident Worker rollover takes to settle) and a
|
|
7730
|
-
* timeout equal to the method's own budget (never above the engine's 30
|
|
7731
|
-
* minutes). Inputs to a step are the event's `resource`, the instance id and
|
|
7732
|
-
* previous steps' returns — refs, shas, a key, a path — never a payload and
|
|
7733
|
-
* never a credential. A step killed from outside (the container replaced
|
|
7734
|
-
* under it) throws and the engine retries it into the same idempotent
|
|
7735
|
-
* method; a gate that ends the cycle (idle, a container restart) or a
|
|
7736
|
-
* failure of the repository's own (recorded as `degraded`, the last snapshot
|
|
7737
|
-
* still serving) ends the instance with that word, and the next cron firing
|
|
7738
|
-
* creates the next one from the row's state. The instance runs one cycle
|
|
7739
|
-
* and returns: it is created by the watchdog cron per resident and
|
|
7740
|
-
* ten-minute bucket (`createRefreshInstance`), never a loop.
|
|
7741
|
-
*
|
|
7742
|
-
* The run is the cycle's root span, `resident.refresh` carrying the instance
|
|
7743
|
-
* id (docs/reference/specs/tracing.md item 25), with every command a step ran
|
|
7744
|
-
* grafted under it as a `resident.<step>` child — the same shape the alarm's
|
|
7745
|
-
* root has. */
|
|
7746
|
-
export class ResidentRefresh extends WorkflowEntrypoint<Env, RefreshInstanceParams> {
|
|
7747
|
-
async run(
|
|
7748
|
-
event: Readonly<WorkflowEvent<RefreshInstanceParams>>,
|
|
7749
|
-
step: WorkflowStep,
|
|
7750
|
-
): Promise<RefreshInstanceSummary> {
|
|
7751
|
-
const { resource } = event.payload;
|
|
7752
|
-
const instance = event.instanceId;
|
|
7753
|
-
const stub = residentStub(this.env, resource);
|
|
7754
|
-
const root = startAdoptedRoot(tracer, "resident.refresh", {
|
|
7755
|
-
sinks: traceSinks,
|
|
7756
|
-
startedAt: event.timestamp.getTime(),
|
|
7757
|
-
attrs: { instanceId: instance },
|
|
7758
|
-
});
|
|
7759
|
-
const graft = (answer: InstanceStepTrace) =>
|
|
7760
|
-
graftResidentSteps(answer.trace, {
|
|
7761
|
-
parent: root,
|
|
7762
|
-
prefix: "resident",
|
|
7763
|
-
baseAt: answer.startedAt,
|
|
7764
|
-
clipAt: systemClock(),
|
|
7765
|
-
});
|
|
7766
|
-
const retries = REFRESH_STEP_RETRIES;
|
|
7767
|
-
/** The word a step that did not finish ends the instance with, and its outcome for the root. */
|
|
7768
|
-
const ended = (
|
|
7769
|
-
at: RefreshInstanceSummary["step"],
|
|
7770
|
-
answer: { status: "stopped"; why: string } | { status: "failed"; reason: string },
|
|
7771
|
-
): RefreshInstanceSummary => ({
|
|
7772
|
-
instance,
|
|
7773
|
-
outcome: answer.status === "stopped" ? answer.why : "failed",
|
|
7774
|
-
step: at,
|
|
7775
|
-
});
|
|
7776
|
-
let summary: RefreshInstanceSummary | undefined;
|
|
7777
|
-
try {
|
|
7778
|
-
const fetched = await step.do("fetch", { retries, timeout: stepTimeoutMs(REFRESH_FETCH_STEP_BUDGET_MS) }, () =>
|
|
7779
|
-
stub.refreshInstanceFetch({ resource, instance }),
|
|
7780
|
-
);
|
|
7781
|
-
graft(fetched);
|
|
7782
|
-
if (fetched.status !== "done") return (summary = ended("fetch", fetched));
|
|
7783
|
-
let depsEntry: string | null = null;
|
|
7784
|
-
if (fetched.install) {
|
|
7785
|
-
const installed = await step.do(
|
|
7786
|
-
"install",
|
|
7787
|
-
{ retries, timeout: stepTimeoutMs(REFRESH_INSTALL_STEP_BUDGET_MS) },
|
|
7788
|
-
() => stub.refreshInstanceInstall({ resource, instance, sha: fetched.sha, lockfileKey: fetched.lockfileKey }),
|
|
7789
|
-
);
|
|
7790
|
-
graft(installed);
|
|
7791
|
-
if (installed.status !== "done") return (summary = ended("install", installed));
|
|
7792
|
-
depsEntry = installed.entry;
|
|
7793
|
-
}
|
|
7794
|
-
if (fetched.action !== "unchanged") {
|
|
7795
|
-
const built = await step.do("build", { retries, timeout: stepTimeoutMs(REFRESH_BUILD_STEP_BUDGET_MS) }, () =>
|
|
7796
|
-
stub.refreshInstanceBuild({
|
|
7797
|
-
resource,
|
|
7798
|
-
instance,
|
|
7799
|
-
sha: fetched.sha,
|
|
7800
|
-
factsSha: fetched.factsSha,
|
|
7801
|
-
lockfileKey: fetched.lockfileKey,
|
|
7802
|
-
depsEntry,
|
|
7803
|
-
}),
|
|
7804
|
-
);
|
|
7805
|
-
graft(built);
|
|
7806
|
-
if (built.status !== "done") return (summary = ended("build", built));
|
|
7807
|
-
}
|
|
7808
|
-
const snapped = await step.do(
|
|
7809
|
-
"snapshot",
|
|
7810
|
-
{ retries, timeout: stepTimeoutMs(REFRESH_SNAPSHOT_STEP_BUDGET_MS) },
|
|
7811
|
-
() =>
|
|
7812
|
-
stub.refreshInstanceSnapshot({
|
|
7813
|
-
resource,
|
|
7814
|
-
instance,
|
|
7815
|
-
ref: fetched.ref,
|
|
7816
|
-
sha: fetched.sha,
|
|
7817
|
-
lockfileKey: fetched.lockfileKey,
|
|
7818
|
-
action: fetched.action,
|
|
7819
|
-
mintError: fetched.mintError,
|
|
7820
|
-
}),
|
|
7821
|
-
);
|
|
7822
|
-
graft(snapped);
|
|
7823
|
-
if (snapped.status !== "done") return (summary = ended("snapshot", snapped));
|
|
7824
|
-
return (summary = { instance, outcome: "ok", step: "snapshot", action: fetched.action, sha: fetched.sha });
|
|
7825
|
-
} catch (err) {
|
|
7826
|
-
// A step out of retries: the engine records the failed instance by id;
|
|
7827
|
-
// the row keeps its last state and the next cron firing starts the next cycle.
|
|
7828
|
-
root.fail(err);
|
|
7829
|
-
throw err;
|
|
7830
|
-
} finally {
|
|
7831
|
-
const outcome = summary?.outcome ?? "error";
|
|
7832
|
-
root.end(outcome === "ok" || (summary !== undefined && outcome !== "failed") ? "ok" : "error", { outcome });
|
|
7833
|
-
}
|
|
7834
|
-
}
|
|
7835
|
-
}
|
|
7836
|
-
|
|
7837
7619
|
function json(data: unknown, status = 200): Response {
|
|
7838
7620
|
// Item 62: every non-streamed body leaves through here; its `error`,
|
|
7839
7621
|
// `reason` and `summary` strings are made safe at the exit.
|
|
@@ -111,7 +111,8 @@
|
|
|
111
111
|
// account never share one — and BACKUP_BUCKET_NAME must say the same name.
|
|
112
112
|
"r2_buckets": [{ "binding": "BACKUP_BUCKET", "bucket_name": "{{script}}-cache" }],
|
|
113
113
|
// The refresh cycle as a Workflow instance (docs/reference/specs/resident-repos.md
|
|
114
|
-
// item 7): `ResidentRefresh` in
|
|
114
|
+
// item 7): `ResidentRefresh` in refresh.ts, re-exported by worker.ts (the
|
|
115
|
+
// entry `class_name` resolves against), runs one cycle — fetch, install,
|
|
115
116
|
// build, snapshot — as steps the engine retries and records, calling the
|
|
116
117
|
// resident's own step methods. The watchdog cron below creates one instance
|
|
117
118
|
// per resident whose row says `lifecycle: workflow` (the admin `/debug`
|