@coreplane/switchboard 1.202.0 → 1.203.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/switchboard.js +26 -0
- package/dist/assets/config/config.example.yaml +10 -0
- package/dist/assets/deploy/cloudflare-resident/preflight.mjs +3 -3
- package/dist/assets/deploy/cloudflare-resident/refresh.ts +179 -93
- package/dist/assets/deploy/cloudflare-resident/shared.ts +15 -10
- package/dist/assets/deploy/cloudflare-resident/worker.ts +459 -632
- package/dist/assets/package-lock.json +4 -4
- package/dist/assets/package.json +1 -1
- package/dist/assets/source.json +3 -3
- package/dist/assets/src/core/authz/policy.ts +4 -0
- package/dist/assets/src/core/runRecord.ts +11 -0
- package/dist/assets/src/core/schedules.ts +12 -6
- package/dist/assets/src/core/ship/handoff.ts +246 -0
- package/dist/assets/src/execution/residentDisk.ts +2 -2
- package/dist/assets/src/execution/residentInstanceId.ts +48 -45
- package/dist/assets/src/execution/residentRefresh.ts +17 -66
- package/dist/assets/src/execution/residentState.ts +8 -9
- package/dist/assets/src/execution/residentStepPlan.ts +1 -1
- package/dist/assets/web/dist/.vite/manifest.json +55 -44
- package/dist/assets/web/dist/assets/{AppShell-CJLPO_aI.js → AppShell-Bw3-_TTE.js} +1 -1
- package/dist/assets/web/dist/assets/{CostsPage-C6ZvnAmH.js → CostsPage-XQ-fQzdz.js} +1 -1
- package/dist/assets/web/dist/assets/DeliveryPage-BRwcQyr7.js +1 -0
- package/dist/assets/web/dist/assets/{NotFoundPage-BdLcY60r.js → NotFoundPage-RjH-9dPy.js} +1 -1
- package/dist/assets/web/dist/assets/{ResidentDetailPage-hXWrr8T6.js → ResidentDetailPage-BRy5wkv9.js} +1 -1
- package/dist/assets/web/dist/assets/{ResidentsIndexPage-YLu4JwxJ.js → ResidentsIndexPage-DAEVLN8j.js} +1 -1
- package/dist/assets/web/dist/assets/{RunRoutePage-BDhHRJVO.js → RunRoutePage-B1KHmkZ9.js} +1 -1
- package/dist/assets/web/dist/assets/{RunsIndexPage-D7zZ5a0l.js → RunsIndexPage-3hUWFpFV.js} +1 -1
- package/dist/assets/web/dist/assets/{RunsTabs-BOUlSa2W.js → RunsTabs-Qn_5TVJU.js} +1 -1
- package/dist/assets/web/dist/assets/{ScheduledPage-BciOQcC-.js → ScheduledPage-DysVvVm1.js} +1 -1
- package/dist/assets/web/dist/assets/{StatusDot-CTDLBv92.js → StatusDot-Bug2a6T6.js} +1 -1
- package/dist/assets/web/dist/assets/{Tooltip-DD2v9Gxx.js → Tooltip-C2eEUbwn.js} +1 -1
- package/dist/assets/web/dist/assets/{favicon-C4Q8cRsP.js → favicon-CFfbl2AI.js} +1 -1
- package/dist/assets/web/dist/assets/{main-DrSlUcMg.js → main-0uhp-taL.js} +3 -3
- package/dist/assets/web/dist/assets/main-Xickfv8V.css +1 -0
- package/dist/cli.js +2932 -1875
- package/package.json +3 -2
- package/dist/assets/web/dist/assets/main-DoTjwE-G.css +0 -1
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// The `switchboard` bin: this file hands the process to dist/cli.js, the bundle
|
|
3
|
+
// the build writes (build.mts; docs/reference/specs/packaging.md item 1). A
|
|
4
|
+
// committed entry rather than the bundle itself, because npm links a bin only
|
|
5
|
+
// when its target exists at install time, and the gitignored dist/ does not
|
|
6
|
+
// until the build runs: with dist/cli.js as the bin, `npx <the package>` inside
|
|
7
|
+
// the checkout — where npx prefers the workspace over the registry — found no
|
|
8
|
+
// link and died with `sh: switchboard: command not found`. The published package
|
|
9
|
+
// always carries the bundle, so there this is one extra module load; a checkout
|
|
10
|
+
// that has not built the package is told what to run.
|
|
11
|
+
import { existsSync } from "node:fs";
|
|
12
|
+
import { fileURLToPath } from "node:url";
|
|
13
|
+
|
|
14
|
+
const bundle = new URL("../dist/cli.js", import.meta.url);
|
|
15
|
+
|
|
16
|
+
if (!existsSync(bundle)) {
|
|
17
|
+
console.error(
|
|
18
|
+
`switchboard: ${fileURLToPath(bundle)} is missing — the package is not built. Inside the checkout, npx runs this workspace, not the published package: run \`npm run build -w packages/switchboard\` first, or use the checkout's CLI, \`npm run cli -- <group> <verb> …\`.`,
|
|
19
|
+
);
|
|
20
|
+
process.exit(1);
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
// From here the bundle is the script: its entry claim (src/invokedAsScript.ts)
|
|
24
|
+
// and its usage spelling (`programName`, src/cli.ts) read argv[1].
|
|
25
|
+
process.argv[1] = fileURLToPath(bundle);
|
|
26
|
+
await import(bundle.href);
|
|
@@ -272,6 +272,16 @@ workspaceDir: ./workspaces
|
|
|
272
272
|
# some-other-bucket: uploads
|
|
273
273
|
# anthropicWorkspaceId: wrkspc_... # this app's Anthropic workspace
|
|
274
274
|
|
|
275
|
+
# Delivery indicators: GET /delivery and `delivery report` (docs/reference/specs/delivery.md).
|
|
276
|
+
# Optional. Needs a GitHub credential (the App triple or GH_TOKEN); read-only.
|
|
277
|
+
# delivery:
|
|
278
|
+
# repos: [acme/api, acme/web] # the repositories the page serves; the first is its default
|
|
279
|
+
# # (`delivery report --repo owner/name` needs no entry here)
|
|
280
|
+
# reviewers: [] # logins whose reviews count as the review agent's verdicts,
|
|
281
|
+
# # beside this installation's own App identity
|
|
282
|
+
# agentLogins: [] # logins that are agents beyond the `[bot]` suffix
|
|
283
|
+
# agentCoauthors: [Claude] # `Co-Authored-By:` names that mark a commit as an agent's
|
|
284
|
+
|
|
275
285
|
# Dashboard authentication (docs/reference/specs/access-gate.md). Everything the dashboard
|
|
276
286
|
# serves — /runs*, /residents*, /costs*, /mcp/connect/*, and every /api/* command
|
|
277
287
|
# — sits behind ONE identity gate; `auth` picks which credential it checks:
|
|
@@ -41,8 +41,8 @@ const SETTLED_STATES = new Set(["warm", "degraded", "down"]);
|
|
|
41
41
|
* is `provision-failed`, `down`, and only a rebuild recovers it. */
|
|
42
42
|
const PROVISIONING_STATES = new Set(["onboarding"]);
|
|
43
43
|
/** The engine is executing and an isolate swap merely INTERRUPTS it: a refresh
|
|
44
|
-
*
|
|
45
|
-
* items 44 and 48); a restore is retried by the next hydrate, which first
|
|
44
|
+
* step is retried by its Workflow instance and resumes from its disk
|
|
45
|
+
* checkpoints (resident-repos items 44 and 48); a restore is retried by the next hydrate, which first
|
|
46
46
|
* unmounts and removes whatever the interrupted one left (item 61). These
|
|
47
47
|
* used to refuse like `onboarding`, and a release deploy once spent its whole
|
|
48
48
|
* budget behind a resident stuck in `restoring` — a state an isolate swap
|
|
@@ -162,7 +162,7 @@ export function decide(fetched, { force = false } = {}) {
|
|
|
162
162
|
// Interruptible cycles never refuse; they are named so the log says what the
|
|
163
163
|
// swap interrupts and why that is fine.
|
|
164
164
|
const warningText = interrupting.length
|
|
165
|
-
? ` — WARNING: mid-cycle: ${interrupting.map((m) => `${m.resource} (${m.state})`).join(", ")} — the isolate swap interrupts it;
|
|
165
|
+
? ` — WARNING: mid-cycle: ${interrupting.map((m) => `${m.resource} (${m.state})`).join(", ")} — the isolate swap interrupts it; the refresh instance retries the step from its checkpoints, a restore is retried by the next hydrate (resident-repos items 44/61)`
|
|
166
166
|
: "";
|
|
167
167
|
|
|
168
168
|
if (problems.length === 0) {
|
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
// The refresh cycle as a Workflow instance (docs/reference/specs/resident-repos.md
|
|
2
2
|
// item 7): the `ResidentRefresh` entrypoint the Workflows binding names, the
|
|
3
|
-
// watchdog cron's instance-creation duty (`createRefreshInstance`)
|
|
4
|
-
//
|
|
5
|
-
//
|
|
6
|
-
//
|
|
7
|
-
//
|
|
3
|
+
// watchdog cron's instance-creation duty (`createRefreshInstance`), the admin
|
|
4
|
+
// `refresh-now` op's on-demand creation (`createRefreshInstanceNow`) and the
|
|
5
|
+
// existence probe a duplicate-id refusal is confirmed with. The steps' own
|
|
6
|
+
// work stays in worker.ts as the Durable Object's `refreshInstance*` methods:
|
|
7
|
+
// this module only sequences them through the DO stub. The entry re-exports
|
|
8
|
+
// the class (the binding resolves `class_name` against the entry module), and
|
|
8
9
|
// nothing here imports the entry at runtime — the shared pieces come from
|
|
9
10
|
// shared.ts, the entry's types `type`-only.
|
|
10
11
|
import { WorkflowEntrypoint, type WorkflowEvent, type WorkflowStep } from "cloudflare:workers";
|
|
@@ -12,6 +13,7 @@ import { systemClock } from "../../src/core/trace/clock.js";
|
|
|
12
13
|
import { startAdoptedRoot } from "../../src/core/trace/workerTrace.js";
|
|
13
14
|
import {
|
|
14
15
|
REFRESH_STEP_RETRIES,
|
|
16
|
+
refreshCycleBlocked,
|
|
15
17
|
refreshInstanceId,
|
|
16
18
|
shouldCreateRefreshInstance,
|
|
17
19
|
stepTimeoutMs,
|
|
@@ -21,6 +23,7 @@ import { RESTORE_MAX_MS, type RefreshPlan } from "../../src/execution/residentRe
|
|
|
21
23
|
import { residentText } from "../../src/execution/residentText.js";
|
|
22
24
|
import { graftResidentSteps } from "../../src/execution/residentTrace.js";
|
|
23
25
|
import {
|
|
26
|
+
DEFAULT_EXEC_TIMEOUT_MS,
|
|
24
27
|
DEPS_STEP_OVERHEAD_MS,
|
|
25
28
|
errMsg,
|
|
26
29
|
GIT_NETWORK_TIMEOUT_MS,
|
|
@@ -30,6 +33,7 @@ import {
|
|
|
30
33
|
REFRESH_INSTALL_TIMEOUT_MS,
|
|
31
34
|
REFRESH_INTERVAL_S,
|
|
32
35
|
residentStub,
|
|
36
|
+
THREAD_POOL_SIZE,
|
|
33
37
|
tracer,
|
|
34
38
|
traceSinks,
|
|
35
39
|
} from "./shared";
|
|
@@ -41,7 +45,7 @@ export interface RefreshInstanceParams {
|
|
|
41
45
|
resource: string;
|
|
42
46
|
}
|
|
43
47
|
|
|
44
|
-
/** What
|
|
48
|
+
/** What a creator did about one resident's refresh instance. */
|
|
45
49
|
export interface RefreshInstanceAction {
|
|
46
50
|
id: string;
|
|
47
51
|
action: "created" | "duplicate" | "skipped" | "failed";
|
|
@@ -54,7 +58,13 @@ export interface RefreshInstanceAction {
|
|
|
54
58
|
const REFRESH_FETCH_STEP_BUDGET_MS = RESTORE_MAX_MS + GIT_NETWORK_TIMEOUT_MS; // a wake's restore, then the fetch
|
|
55
59
|
const REFRESH_INSTALL_STEP_BUDGET_MS = REFRESH_INSTALL_TIMEOUT_MS + DEPS_STEP_OVERHEAD_MS; // the install's own lease
|
|
56
60
|
const REFRESH_BUILD_STEP_BUDGET_MS = GIT_NETWORK_TIMEOUT_MS + REFRESH_BUILD_TIMEOUT_MS; // the build's mutex lease
|
|
57
|
-
const REFRESH_SNAPSHOT_STEP_BUDGET_MS = R2_TRANSFER_TIMEOUT_MS + GIT_NETWORK_TIMEOUT_MS; // the archives, then the reclaim pass
|
|
61
|
+
const REFRESH_SNAPSHOT_STEP_BUDGET_MS = R2_TRANSFER_TIMEOUT_MS + GIT_NETWORK_TIMEOUT_MS; // the archives, then the reclaim pass
|
|
62
|
+
/** The sweep: one cleanliness check per binding idle past an hour (at most the
|
|
63
|
+
* pool's worth, one exec budget each), then the removals under the mirror lock. */
|
|
64
|
+
const REFRESH_SWEEP_STEP_BUDGET_MS = THREAD_POOL_SIZE * DEFAULT_EXEC_TIMEOUT_MS + GIT_NETWORK_TIMEOUT_MS;
|
|
65
|
+
/** The measurement: `df`, the `du` over the parts and the store's listing (one
|
|
66
|
+
* network-class budget each), then the store's removals. */
|
|
67
|
+
const REFRESH_MEASURE_STEP_BUDGET_MS = 2 * GIT_NETWORK_TIMEOUT_MS + DEFAULT_EXEC_TIMEOUT_MS;
|
|
58
68
|
|
|
59
69
|
/** Whether the engine knows an instance by this id, in any status. A missing
|
|
60
70
|
* id rejects on `get` or on `status`; either way the answer is false. */
|
|
@@ -67,51 +77,85 @@ async function refreshInstanceExists(env: Env, id: string): Promise<boolean> {
|
|
|
67
77
|
}
|
|
68
78
|
}
|
|
69
79
|
|
|
70
|
-
/** The
|
|
71
|
-
|
|
72
|
-
|
|
80
|
+
/** The deterministic id for this resident's current bucket. */
|
|
81
|
+
function bucketInstanceId(resource: string, nowMs: number): string {
|
|
82
|
+
const slug = resource.slice("repo:".length);
|
|
83
|
+
const slash = slug.indexOf("/");
|
|
84
|
+
return refreshInstanceId(slug.slice(0, slash), slug.slice(slash + 1), nowMs);
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/** Create the instance `id` for `resource` and record the outcome on the row.
|
|
88
|
+
* The engine refuses an id that names an instance still inside its retention,
|
|
89
|
+
* and the refusal carries no code — so the id is asked, not the wording: an
|
|
90
|
+
* instance that answers for it exists, and the refusal was the duplicate it
|
|
91
|
+
* looks like. Any other failure stays a failure. */
|
|
92
|
+
async function createInstance(
|
|
93
|
+
env: Env,
|
|
94
|
+
stub: ReturnType<typeof residentStub>,
|
|
95
|
+
resource: string,
|
|
96
|
+
id: string,
|
|
97
|
+
nowMs: number,
|
|
98
|
+
why: string,
|
|
99
|
+
): Promise<RefreshInstanceAction> {
|
|
100
|
+
try {
|
|
101
|
+
await env.RESIDENT_REFRESH.create({ id, params: { resource } });
|
|
102
|
+
} catch (err) {
|
|
103
|
+
const message = errMsg(err);
|
|
104
|
+
if (await refreshInstanceExists(env, id)) {
|
|
105
|
+
await stub.recordRefreshSkipped(id, nowMs, "duplicate");
|
|
106
|
+
return { id, action: "duplicate", why: "duplicate" };
|
|
107
|
+
}
|
|
108
|
+
console.error(`resident-watchdog: creating refresh instance ${id} failed — ${message}`);
|
|
109
|
+
return { id, action: "failed", why: residentText(message) };
|
|
110
|
+
}
|
|
111
|
+
await stub.recordRefreshInstance(id, nowMs);
|
|
112
|
+
console.log(`resident-watchdog: created refresh instance ${id} (${why})`);
|
|
113
|
+
return { id, action: "created", why };
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/** The cron's instance-creation duty for one resident (item 7): at most one
|
|
117
|
+
* instance per ten-minute bucket, never while a cycle is live, only once the
|
|
118
|
+
* cadence has elapsed (`shouldCreateRefreshInstance`); the id is
|
|
73
119
|
* deterministic per resident and bucket, so a second firing in one bucket
|
|
74
120
|
* meets the engine's duplicate-id refusal, which is the expected no-op. A
|
|
75
|
-
* skipped live cycle and a duplicate are recorded on the row for `/status`.
|
|
76
|
-
* An `alarm` resident answers null: nothing here touches it. */
|
|
121
|
+
* skipped live cycle and a duplicate are recorded on the row for `/status`. */
|
|
77
122
|
export async function createRefreshInstance(
|
|
78
123
|
env: Env,
|
|
79
124
|
stub: ReturnType<typeof residentStub>,
|
|
80
125
|
resource: string,
|
|
81
126
|
row: RefreshRow,
|
|
82
|
-
): Promise<RefreshInstanceAction
|
|
83
|
-
if (row.lifecycle !== "workflow") return null;
|
|
127
|
+
): Promise<RefreshInstanceAction> {
|
|
84
128
|
const now = systemClock();
|
|
85
129
|
const decision = shouldCreateRefreshInstance(row, now, {
|
|
86
130
|
intervalS: REFRESH_INTERVAL_S,
|
|
87
131
|
idleIntervalS: IDLE_REFRESH_INTERVAL_S,
|
|
88
132
|
});
|
|
89
|
-
const
|
|
90
|
-
const slash = slug.indexOf("/");
|
|
91
|
-
const id = refreshInstanceId(slug.slice(0, slash), slug.slice(slash + 1), now);
|
|
133
|
+
const id = bucketInstanceId(resource, now);
|
|
92
134
|
if (!decision.create) {
|
|
93
135
|
if (decision.why === "mid-cycle" || decision.why === "running")
|
|
94
136
|
await stub.recordRefreshSkipped(id, now, decision.why);
|
|
95
137
|
return { id, action: "skipped", why: decision.why };
|
|
96
138
|
}
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
139
|
+
return createInstance(env, stub, resource, id, now, decision.why);
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
/** The admin `refresh-now` op (item 13): this bucket's instance, created now,
|
|
143
|
+
* due or not — the cycle the cron would create at its next firing, started
|
|
144
|
+
* on demand. Never beside a live cycle (`refreshCycleBlocked`: a running
|
|
145
|
+
* instance, a young `refreshing` marker, a resident that is not serving —
|
|
146
|
+
* answered `skipped` with the why), and under the bucket's deterministic id,
|
|
147
|
+
* so a bucket the cron already served answers `duplicate` naming that
|
|
148
|
+
* instance: the next bucket is at most ten minutes away. */
|
|
149
|
+
export async function createRefreshInstanceNow(
|
|
150
|
+
env: Env,
|
|
151
|
+
stub: ReturnType<typeof residentStub>,
|
|
152
|
+
resource: string,
|
|
153
|
+
): Promise<RefreshInstanceAction> {
|
|
154
|
+
const now = systemClock();
|
|
155
|
+
const id = bucketInstanceId(resource, now);
|
|
156
|
+
const blocked = refreshCycleBlocked(await stub.refreshRow(), now);
|
|
157
|
+
if (blocked) return { id, action: "skipped", why: blocked };
|
|
158
|
+
return createInstance(env, stub, resource, id, now, "refresh-now");
|
|
115
159
|
}
|
|
116
160
|
|
|
117
161
|
/** What an instance answers when it ends: small facts for the engine's record. */
|
|
@@ -119,33 +163,42 @@ interface RefreshInstanceSummary {
|
|
|
119
163
|
instance: string;
|
|
120
164
|
/** `ok`, or the word a gate or a failure ended the cycle with. */
|
|
121
165
|
outcome: string;
|
|
166
|
+
/** The step the cycle's verdict came from. */
|
|
122
167
|
step: "fetch" | "install" | "build" | "snapshot";
|
|
123
168
|
action?: RefreshPlan["action"];
|
|
124
169
|
sha?: string;
|
|
170
|
+
/** The housekeeping steps' answers, when they ran. */
|
|
171
|
+
swept?: { evicted: number; kept: number } | null;
|
|
172
|
+
measured?: boolean;
|
|
125
173
|
}
|
|
126
174
|
|
|
127
175
|
/** One refresh cycle as one short Workflow instance: `fetch`, `install` (only
|
|
128
176
|
* when the plan moved the lockfile key), `build` (only when the branch
|
|
129
|
-
* moved), `snapshot
|
|
130
|
-
*
|
|
131
|
-
*
|
|
132
|
-
*
|
|
133
|
-
*
|
|
134
|
-
*
|
|
135
|
-
*
|
|
136
|
-
*
|
|
137
|
-
*
|
|
138
|
-
*
|
|
139
|
-
*
|
|
140
|
-
*
|
|
141
|
-
*
|
|
142
|
-
*
|
|
143
|
-
*
|
|
177
|
+
* moved), `snapshot`, then the housekeeping `sweep` (the worktree
|
|
178
|
+
* inactivity eviction, item 23) and `measure` (the disk sample, item 55) —
|
|
179
|
+
* each a `step.do` calling the resident's own step method through the DO
|
|
180
|
+
* stub, under the retry policy `REFRESH_STEP_RETRIES` (six attempts, thirty
|
|
181
|
+
* seconds apart, doubling: about 15.5 minutes, past the 3 to 10 minutes a
|
|
182
|
+
* resident Worker rollover takes to settle) and a timeout equal to the
|
|
183
|
+
* method's own budget (never above the engine's 30 minutes). Inputs to a
|
|
184
|
+
* step are the event's `resource`, the instance id and previous steps'
|
|
185
|
+
* returns — refs, shas, a key, a path — never a payload and never a
|
|
186
|
+
* credential. A step killed from outside (the container replaced under it)
|
|
187
|
+
* throws and the engine retries it into the same idempotent method; so does
|
|
188
|
+
* a gate that stopped the container on purpose (a stale image, a disk-full
|
|
189
|
+
* recycle) — the retry finds the container back and continues the cycle. A
|
|
190
|
+
* gate that ends the cycle (idle, an offboard) or a failure of the
|
|
191
|
+
* repository's own (recorded as `degraded`, the last snapshot still serving)
|
|
192
|
+
* ends the cycle with that word; the housekeeping steps still run for every
|
|
193
|
+
* verdict but an offboard (nothing is left to keep), and neither wakes a
|
|
194
|
+
* slept container. The instance runs one cycle and returns: it is created by
|
|
195
|
+
* the watchdog cron per resident and ten-minute bucket
|
|
196
|
+
* (`createRefreshInstance`) or by the admin `refresh-now` op, never a loop;
|
|
197
|
+
* the next cron firing creates the next one from the row's state.
|
|
144
198
|
*
|
|
145
199
|
* The run is the cycle's root span, `resident.refresh` carrying the instance
|
|
146
200
|
* id (docs/reference/specs/tracing.md item 25), with every command a step ran
|
|
147
|
-
* grafted under it as a `resident.<step>` child
|
|
148
|
-
* root has. */
|
|
201
|
+
* grafted under it as a `resident.<step>` child. */
|
|
149
202
|
export class ResidentRefresh extends WorkflowEntrypoint<Env, RefreshInstanceParams> {
|
|
150
203
|
async run(
|
|
151
204
|
event: Readonly<WorkflowEvent<RefreshInstanceParams>>,
|
|
@@ -167,7 +220,7 @@ export class ResidentRefresh extends WorkflowEntrypoint<Env, RefreshInstancePara
|
|
|
167
220
|
clipAt: systemClock(),
|
|
168
221
|
});
|
|
169
222
|
const retries = REFRESH_STEP_RETRIES;
|
|
170
|
-
/** The word a step that did not finish ends the
|
|
223
|
+
/** The word a step that did not finish ends the cycle with. */
|
|
171
224
|
const ended = (
|
|
172
225
|
at: RefreshInstanceSummary["step"],
|
|
173
226
|
answer: { status: "stopped"; why: string } | { status: "failed"; reason: string },
|
|
@@ -178,53 +231,86 @@ export class ResidentRefresh extends WorkflowEntrypoint<Env, RefreshInstancePara
|
|
|
178
231
|
});
|
|
179
232
|
let summary: RefreshInstanceSummary | undefined;
|
|
180
233
|
try {
|
|
234
|
+
let cycle: RefreshInstanceSummary | undefined;
|
|
181
235
|
const fetched = await step.do("fetch", { retries, timeout: stepTimeoutMs(REFRESH_FETCH_STEP_BUDGET_MS) }, () =>
|
|
182
236
|
stub.refreshInstanceFetch({ resource, instance }),
|
|
183
237
|
);
|
|
184
238
|
graft(fetched);
|
|
185
|
-
if (fetched.status !== "done")
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
239
|
+
if (fetched.status !== "done") cycle = ended("fetch", fetched);
|
|
240
|
+
else {
|
|
241
|
+
let depsEntry: string | null = null;
|
|
242
|
+
if (fetched.install) {
|
|
243
|
+
const installed = await step.do(
|
|
244
|
+
"install",
|
|
245
|
+
{ retries, timeout: stepTimeoutMs(REFRESH_INSTALL_STEP_BUDGET_MS) },
|
|
246
|
+
() =>
|
|
247
|
+
stub.refreshInstanceInstall({ resource, instance, sha: fetched.sha, lockfileKey: fetched.lockfileKey }),
|
|
248
|
+
);
|
|
249
|
+
graft(installed);
|
|
250
|
+
if (installed.status !== "done") cycle = ended("install", installed);
|
|
251
|
+
else depsEntry = installed.entry;
|
|
252
|
+
}
|
|
253
|
+
if (cycle === undefined && fetched.action !== "unchanged") {
|
|
254
|
+
const built = await step.do("build", { retries, timeout: stepTimeoutMs(REFRESH_BUILD_STEP_BUDGET_MS) }, () =>
|
|
255
|
+
stub.refreshInstanceBuild({
|
|
256
|
+
resource,
|
|
257
|
+
instance,
|
|
258
|
+
sha: fetched.sha,
|
|
259
|
+
factsSha: fetched.factsSha,
|
|
260
|
+
lockfileKey: fetched.lockfileKey,
|
|
261
|
+
depsEntry,
|
|
262
|
+
}),
|
|
263
|
+
);
|
|
264
|
+
graft(built);
|
|
265
|
+
if (built.status !== "done") cycle = ended("build", built);
|
|
266
|
+
}
|
|
267
|
+
if (cycle === undefined) {
|
|
268
|
+
const snapped = await step.do(
|
|
269
|
+
"snapshot",
|
|
270
|
+
{ retries, timeout: stepTimeoutMs(REFRESH_SNAPSHOT_STEP_BUDGET_MS) },
|
|
271
|
+
() =>
|
|
272
|
+
stub.refreshInstanceSnapshot({
|
|
273
|
+
resource,
|
|
274
|
+
instance,
|
|
275
|
+
ref: fetched.ref,
|
|
276
|
+
sha: fetched.sha,
|
|
277
|
+
lockfileKey: fetched.lockfileKey,
|
|
278
|
+
action: fetched.action,
|
|
279
|
+
mintError: fetched.mintError,
|
|
280
|
+
}),
|
|
281
|
+
);
|
|
282
|
+
graft(snapped);
|
|
283
|
+
cycle =
|
|
284
|
+
snapped.status !== "done"
|
|
285
|
+
? ended("snapshot", snapped)
|
|
286
|
+
: { instance, outcome: "ok", step: "snapshot", action: fetched.action, sha: fetched.sha };
|
|
287
|
+
}
|
|
196
288
|
}
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
lockfileKey: fetched.lockfileKey,
|
|
205
|
-
depsEntry,
|
|
206
|
-
}),
|
|
289
|
+
// Housekeeping rides on every instance whatever the cycle's verdict — a
|
|
290
|
+
// parked resident still releases its idle bindings and re-measures, and
|
|
291
|
+
// neither step wakes a slept container — except an offboarded one:
|
|
292
|
+
// nothing is left to keep.
|
|
293
|
+
if (cycle.outcome !== "offboarded") {
|
|
294
|
+
const swept = await step.do("sweep", { retries, timeout: stepTimeoutMs(REFRESH_SWEEP_STEP_BUDGET_MS) }, () =>
|
|
295
|
+
stub.refreshInstanceSweep({ resource, instance }),
|
|
207
296
|
);
|
|
208
|
-
graft(
|
|
209
|
-
|
|
297
|
+
graft(swept);
|
|
298
|
+
const measured = await step.do(
|
|
299
|
+
"measure",
|
|
300
|
+
{ retries, timeout: stepTimeoutMs(REFRESH_MEASURE_STEP_BUDGET_MS) },
|
|
301
|
+
() => stub.refreshInstanceMeasure({ instance }),
|
|
302
|
+
);
|
|
303
|
+
graft(measured);
|
|
304
|
+
cycle = {
|
|
305
|
+
...cycle,
|
|
306
|
+
swept:
|
|
307
|
+
swept.status === "done" && swept.result
|
|
308
|
+
? { evicted: swept.result.evicted.length, kept: swept.result.kept }
|
|
309
|
+
: null,
|
|
310
|
+
measured: measured.status === "done" && measured.result !== null && measured.result.measured,
|
|
311
|
+
};
|
|
210
312
|
}
|
|
211
|
-
|
|
212
|
-
"snapshot",
|
|
213
|
-
{ retries, timeout: stepTimeoutMs(REFRESH_SNAPSHOT_STEP_BUDGET_MS) },
|
|
214
|
-
() =>
|
|
215
|
-
stub.refreshInstanceSnapshot({
|
|
216
|
-
resource,
|
|
217
|
-
instance,
|
|
218
|
-
ref: fetched.ref,
|
|
219
|
-
sha: fetched.sha,
|
|
220
|
-
lockfileKey: fetched.lockfileKey,
|
|
221
|
-
action: fetched.action,
|
|
222
|
-
mintError: fetched.mintError,
|
|
223
|
-
}),
|
|
224
|
-
);
|
|
225
|
-
graft(snapped);
|
|
226
|
-
if (snapped.status !== "done") return (summary = ended("snapshot", snapped));
|
|
227
|
-
return (summary = { instance, outcome: "ok", step: "snapshot", action: fetched.action, sha: fetched.sha });
|
|
313
|
+
return (summary = cycle);
|
|
228
314
|
} catch (err) {
|
|
229
315
|
// A step out of retries: the engine records the failed instance by id;
|
|
230
316
|
// the row keeps its last state and the next cron firing starts the next cycle.
|
|
@@ -19,20 +19,25 @@ import type { Env } from "./worker";
|
|
|
19
19
|
* resident is re-warmed before the platform can sleep it. Bump together. */
|
|
20
20
|
export const SLEEP_AFTER = "20m";
|
|
21
21
|
|
|
22
|
-
/**
|
|
23
|
-
*
|
|
24
|
-
*
|
|
25
|
-
*
|
|
26
|
-
* within one refresh interval. */
|
|
22
|
+
/** The refresh cadence (seconds): one instance per resident per ten-minute
|
|
23
|
+
* bucket, created by the watchdog cron, so the cron's own period is the
|
|
24
|
+
* cadence. The cycle doubles as the keep-warm heartbeat (its steps call into
|
|
25
|
+
* the container), so it must stay below SLEEP_AFTER. */
|
|
27
26
|
export const REFRESH_INTERVAL_S = 600;
|
|
28
27
|
|
|
29
|
-
/** How far out an idle resident's cycle is
|
|
30
|
-
*
|
|
31
|
-
*
|
|
28
|
+
/** How far out an idle resident's next cycle is (seconds): the cron's idle
|
|
29
|
+
* cadence (`IDLE_AFTER_S` in worker.ts decides idleness), in whole
|
|
30
|
+
* ten-minute buckets since the last instance. */
|
|
32
31
|
export const IDLE_REFRESH_INTERVAL_S = 6 * 60 * 60;
|
|
33
32
|
|
|
34
|
-
/**
|
|
35
|
-
*
|
|
33
|
+
/** The thread user pool the image carries (`worker2`..`worker17`): one OS
|
|
34
|
+
* user per attached thread, and the bound on how many bindings one sweep
|
|
35
|
+
* step can have to check. */
|
|
36
|
+
export const THREAD_POOL_SIZE = 16;
|
|
37
|
+
|
|
38
|
+
/** Exec budgets. Each is a refresh step's own budget too (refresh.ts), so a
|
|
39
|
+
* step timeout and the command timeout it bounds agree; every one is under
|
|
40
|
+
* the engine's 30-minute step ceiling. */
|
|
36
41
|
export const DEFAULT_EXEC_TIMEOUT_MS = 60_000;
|
|
37
42
|
export const GIT_NETWORK_TIMEOUT_MS = 5 * 60_000;
|
|
38
43
|
export const REFRESH_BUILD_TIMEOUT_MS = 5 * 60_000;
|