@coreplane/switchboard 1.202.0 → 1.203.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/bin/switchboard.js +26 -0
  2. package/dist/assets/config/config.example.yaml +10 -0
  3. package/dist/assets/deploy/cloudflare-resident/preflight.mjs +3 -3
  4. package/dist/assets/deploy/cloudflare-resident/refresh.ts +179 -93
  5. package/dist/assets/deploy/cloudflare-resident/shared.ts +15 -10
  6. package/dist/assets/deploy/cloudflare-resident/worker.ts +459 -632
  7. package/dist/assets/package-lock.json +4 -4
  8. package/dist/assets/package.json +1 -1
  9. package/dist/assets/source.json +3 -3
  10. package/dist/assets/src/core/authz/policy.ts +4 -0
  11. package/dist/assets/src/core/runRecord.ts +11 -0
  12. package/dist/assets/src/core/schedules.ts +12 -6
  13. package/dist/assets/src/core/ship/handoff.ts +246 -0
  14. package/dist/assets/src/execution/residentDisk.ts +2 -2
  15. package/dist/assets/src/execution/residentInstanceId.ts +48 -45
  16. package/dist/assets/src/execution/residentRefresh.ts +17 -66
  17. package/dist/assets/src/execution/residentState.ts +8 -9
  18. package/dist/assets/src/execution/residentStepPlan.ts +1 -1
  19. package/dist/assets/web/dist/.vite/manifest.json +55 -44
  20. package/dist/assets/web/dist/assets/{AppShell-CJLPO_aI.js → AppShell-Bw3-_TTE.js} +1 -1
  21. package/dist/assets/web/dist/assets/{CostsPage-C6ZvnAmH.js → CostsPage-XQ-fQzdz.js} +1 -1
  22. package/dist/assets/web/dist/assets/DeliveryPage-BRwcQyr7.js +1 -0
  23. package/dist/assets/web/dist/assets/{NotFoundPage-BdLcY60r.js → NotFoundPage-RjH-9dPy.js} +1 -1
  24. package/dist/assets/web/dist/assets/{ResidentDetailPage-hXWrr8T6.js → ResidentDetailPage-BRy5wkv9.js} +1 -1
  25. package/dist/assets/web/dist/assets/{ResidentsIndexPage-YLu4JwxJ.js → ResidentsIndexPage-DAEVLN8j.js} +1 -1
  26. package/dist/assets/web/dist/assets/{RunRoutePage-BDhHRJVO.js → RunRoutePage-B1KHmkZ9.js} +1 -1
  27. package/dist/assets/web/dist/assets/{RunsIndexPage-D7zZ5a0l.js → RunsIndexPage-3hUWFpFV.js} +1 -1
  28. package/dist/assets/web/dist/assets/{RunsTabs-BOUlSa2W.js → RunsTabs-Qn_5TVJU.js} +1 -1
  29. package/dist/assets/web/dist/assets/{ScheduledPage-BciOQcC-.js → ScheduledPage-DysVvVm1.js} +1 -1
  30. package/dist/assets/web/dist/assets/{StatusDot-CTDLBv92.js → StatusDot-Bug2a6T6.js} +1 -1
  31. package/dist/assets/web/dist/assets/{Tooltip-DD2v9Gxx.js → Tooltip-C2eEUbwn.js} +1 -1
  32. package/dist/assets/web/dist/assets/{favicon-C4Q8cRsP.js → favicon-CFfbl2AI.js} +1 -1
  33. package/dist/assets/web/dist/assets/{main-DrSlUcMg.js → main-0uhp-taL.js} +3 -3
  34. package/dist/assets/web/dist/assets/main-Xickfv8V.css +1 -0
  35. package/dist/cli.js +2932 -1875
  36. package/package.json +3 -2
  37. package/dist/assets/web/dist/assets/main-DoTjwE-G.css +0 -1
@@ -0,0 +1,26 @@
1
+ #!/usr/bin/env node
2
+ // The `switchboard` bin: this file hands the process to dist/cli.js, the bundle
3
+ // the build writes (build.mts; docs/reference/specs/packaging.md item 1). A
4
+ // committed entry rather than the bundle itself, because npm links a bin only
5
+ // when its target exists at install time, and the gitignored dist/ does not
6
+ // until the build runs: with dist/cli.js as the bin, `npx <the package>` inside
7
+ // the checkout — where npx prefers the workspace over the registry — found no
8
+ // link and died with `sh: switchboard: command not found`. The published package
9
+ // always carries the bundle, so there this is one extra module load; a checkout
10
+ // that has not built the package is told what to run.
11
+ import { existsSync } from "node:fs";
12
+ import { fileURLToPath } from "node:url";
13
+
14
+ const bundle = new URL("../dist/cli.js", import.meta.url);
15
+
16
+ if (!existsSync(bundle)) {
17
+ console.error(
18
+ `switchboard: ${fileURLToPath(bundle)} is missing — the package is not built. Inside the checkout, npx runs this workspace, not the published package: run \`npm run build -w packages/switchboard\` first, or use the checkout's CLI, \`npm run cli -- <group> <verb> …\`.`,
19
+ );
20
+ process.exit(1);
21
+ }
22
+
23
+ // From here the bundle is the script: its entry claim (src/invokedAsScript.ts)
24
+ // and its usage spelling (`programName`, src/cli.ts) read argv[1].
25
+ process.argv[1] = fileURLToPath(bundle);
26
+ await import(bundle.href);
@@ -272,6 +272,16 @@ workspaceDir: ./workspaces
272
272
  # some-other-bucket: uploads
273
273
  # anthropicWorkspaceId: wrkspc_... # this app's Anthropic workspace
274
274
 
275
+ # Delivery indicators: GET /delivery and `delivery report` (docs/reference/specs/delivery.md).
276
+ # Optional. Needs a GitHub credential (the App triple or GH_TOKEN); read-only.
277
+ # delivery:
278
+ # repos: [acme/api, acme/web] # the repositories the page serves; the first is its default
279
+ # # (`delivery report --repo owner/name` needs no entry here)
280
+ # reviewers: [] # logins whose reviews count as the review agent's verdicts,
281
+ # # beside this installation's own App identity
282
+ # agentLogins: [] # logins that are agents beyond the `[bot]` suffix
283
+ # agentCoauthors: [Claude] # `Co-Authored-By:` names that mark a commit as an agent's
284
+
275
285
  # Dashboard authentication (docs/reference/specs/access-gate.md). Everything the dashboard
276
286
  # serves — /runs*, /residents*, /costs*, /mcp/connect/*, and every /api/* command
277
287
  # — sits behind ONE identity gate; `auth` picks which credential it checks:
@@ -41,8 +41,8 @@ const SETTLED_STATES = new Set(["warm", "degraded", "down"]);
41
41
  * is `provision-failed`, `down`, and only a rebuild recovers it. */
42
42
  const PROVISIONING_STATES = new Set(["onboarding"]);
43
43
  /** The engine is executing and an isolate swap merely INTERRUPTS it: a refresh
44
- * cycle re-arms in 45 s and resumes from its disk checkpoints (resident-repos
45
- * items 44 and 48); a restore is retried by the next hydrate, which first
44
+ * step is retried by its Workflow instance and resumes from its disk
45
+ * checkpoints (resident-repos items 44 and 48); a restore is retried by the next hydrate, which first
46
46
  * unmounts and removes whatever the interrupted one left (item 61). These
47
47
  * used to refuse like `onboarding`, and a release deploy once spent its whole
48
48
  * budget behind a resident stuck in `restoring` — a state an isolate swap
@@ -162,7 +162,7 @@ export function decide(fetched, { force = false } = {}) {
162
162
  // Interruptible cycles never refuse; they are named so the log says what the
163
163
  // swap interrupts and why that is fine.
164
164
  const warningText = interrupting.length
165
- ? ` — WARNING: mid-cycle: ${interrupting.map((m) => `${m.resource} (${m.state})`).join(", ")} — the isolate swap interrupts it; a refresh re-arms in 45 s from its checkpoints, a restore is retried by the next hydrate (resident-repos items 44/61)`
165
+ ? ` — WARNING: mid-cycle: ${interrupting.map((m) => `${m.resource} (${m.state})`).join(", ")} — the isolate swap interrupts it; the refresh instance retries the step from its checkpoints, a restore is retried by the next hydrate (resident-repos items 44/61)`
166
166
  : "";
167
167
 
168
168
  if (problems.length === 0) {
@@ -1,10 +1,11 @@
1
1
  // The refresh cycle as a Workflow instance (docs/reference/specs/resident-repos.md
2
2
  // item 7): the `ResidentRefresh` entrypoint the Workflows binding names, the
3
- // watchdog cron's instance-creation duty (`createRefreshInstance`) and the
4
- // existence probe it confirms a duplicate-id refusal with. The steps' own work
5
- // stays in worker.ts as the Durable Object's `refreshInstance*` methods: this
6
- // module only sequences them through the DO stub. The entry re-exports the
7
- // class (the binding resolves `class_name` against the entry module), and
3
+ // watchdog cron's instance-creation duty (`createRefreshInstance`), the admin
4
+ // `refresh-now` op's on-demand creation (`createRefreshInstanceNow`) and the
5
+ // existence probe a duplicate-id refusal is confirmed with. The steps' own
6
+ // work stays in worker.ts as the Durable Object's `refreshInstance*` methods:
7
+ // this module only sequences them through the DO stub. The entry re-exports
8
+ // the class (the binding resolves `class_name` against the entry module), and
8
9
  // nothing here imports the entry at runtime — the shared pieces come from
9
10
  // shared.ts, the entry's types `type`-only.
10
11
  import { WorkflowEntrypoint, type WorkflowEvent, type WorkflowStep } from "cloudflare:workers";
@@ -12,6 +13,7 @@ import { systemClock } from "../../src/core/trace/clock.js";
12
13
  import { startAdoptedRoot } from "../../src/core/trace/workerTrace.js";
13
14
  import {
14
15
  REFRESH_STEP_RETRIES,
16
+ refreshCycleBlocked,
15
17
  refreshInstanceId,
16
18
  shouldCreateRefreshInstance,
17
19
  stepTimeoutMs,
@@ -21,6 +23,7 @@ import { RESTORE_MAX_MS, type RefreshPlan } from "../../src/execution/residentRe
21
23
  import { residentText } from "../../src/execution/residentText.js";
22
24
  import { graftResidentSteps } from "../../src/execution/residentTrace.js";
23
25
  import {
26
+ DEFAULT_EXEC_TIMEOUT_MS,
24
27
  DEPS_STEP_OVERHEAD_MS,
25
28
  errMsg,
26
29
  GIT_NETWORK_TIMEOUT_MS,
@@ -30,6 +33,7 @@ import {
30
33
  REFRESH_INSTALL_TIMEOUT_MS,
31
34
  REFRESH_INTERVAL_S,
32
35
  residentStub,
36
+ THREAD_POOL_SIZE,
33
37
  tracer,
34
38
  traceSinks,
35
39
  } from "./shared";
@@ -41,7 +45,7 @@ export interface RefreshInstanceParams {
41
45
  resource: string;
42
46
  }
43
47
 
44
- /** What the cron did about one resident's refresh instance this pass. */
48
+ /** What a creator did about one resident's refresh instance. */
45
49
  export interface RefreshInstanceAction {
46
50
  id: string;
47
51
  action: "created" | "duplicate" | "skipped" | "failed";
@@ -54,7 +58,13 @@ export interface RefreshInstanceAction {
54
58
  const REFRESH_FETCH_STEP_BUDGET_MS = RESTORE_MAX_MS + GIT_NETWORK_TIMEOUT_MS; // a wake's restore, then the fetch
55
59
  const REFRESH_INSTALL_STEP_BUDGET_MS = REFRESH_INSTALL_TIMEOUT_MS + DEPS_STEP_OVERHEAD_MS; // the install's own lease
56
60
  const REFRESH_BUILD_STEP_BUDGET_MS = GIT_NETWORK_TIMEOUT_MS + REFRESH_BUILD_TIMEOUT_MS; // the build's mutex lease
57
- const REFRESH_SNAPSHOT_STEP_BUDGET_MS = R2_TRANSFER_TIMEOUT_MS + GIT_NETWORK_TIMEOUT_MS; // the archives, then the reclaim pass and the disk sample
61
+ const REFRESH_SNAPSHOT_STEP_BUDGET_MS = R2_TRANSFER_TIMEOUT_MS + GIT_NETWORK_TIMEOUT_MS; // the archives, then the reclaim pass
62
+ /** The sweep: one cleanliness check per binding idle past an hour (at most the
63
+ * pool's worth, one exec budget each), then the removals under the mirror lock. */
64
+ const REFRESH_SWEEP_STEP_BUDGET_MS = THREAD_POOL_SIZE * DEFAULT_EXEC_TIMEOUT_MS + GIT_NETWORK_TIMEOUT_MS;
65
+ /** The measurement: `df`, the `du` over the parts and the store's listing (one
66
+ * network-class budget each), then the store's removals. */
67
+ const REFRESH_MEASURE_STEP_BUDGET_MS = 2 * GIT_NETWORK_TIMEOUT_MS + DEFAULT_EXEC_TIMEOUT_MS;
58
68
 
59
69
  /** Whether the engine knows an instance by this id, in any status. A missing
60
70
  * id rejects on `get` or on `status`; either way the answer is false. */
@@ -67,51 +77,85 @@ async function refreshInstanceExists(env: Env, id: string): Promise<boolean> {
67
77
  }
68
78
  }
69
79
 
70
- /** The cron's instance-creation duty for one resident (item 7). Only a
71
- * `workflow` row gets an instance, at most one per ten-minute bucket, never
72
- * while a cycle is live (`shouldCreateRefreshInstance`); the id is
80
+ /** The deterministic id for this resident's current bucket. */
81
+ function bucketInstanceId(resource: string, nowMs: number): string {
82
+ const slug = resource.slice("repo:".length);
83
+ const slash = slug.indexOf("/");
84
+ return refreshInstanceId(slug.slice(0, slash), slug.slice(slash + 1), nowMs);
85
+ }
86
+
87
+ /** Create the instance `id` for `resource` and record the outcome on the row.
88
+ * The engine refuses an id that names an instance still inside its retention,
89
+ * and the refusal carries no code — so the id is asked, not the wording: an
90
+ * instance that answers for it exists, and the refusal was the duplicate it
91
+ * looks like. Any other failure stays a failure. */
92
+ async function createInstance(
93
+ env: Env,
94
+ stub: ReturnType<typeof residentStub>,
95
+ resource: string,
96
+ id: string,
97
+ nowMs: number,
98
+ why: string,
99
+ ): Promise<RefreshInstanceAction> {
100
+ try {
101
+ await env.RESIDENT_REFRESH.create({ id, params: { resource } });
102
+ } catch (err) {
103
+ const message = errMsg(err);
104
+ if (await refreshInstanceExists(env, id)) {
105
+ await stub.recordRefreshSkipped(id, nowMs, "duplicate");
106
+ return { id, action: "duplicate", why: "duplicate" };
107
+ }
108
+ console.error(`resident-watchdog: creating refresh instance ${id} failed — ${message}`);
109
+ return { id, action: "failed", why: residentText(message) };
110
+ }
111
+ await stub.recordRefreshInstance(id, nowMs);
112
+ console.log(`resident-watchdog: created refresh instance ${id} (${why})`);
113
+ return { id, action: "created", why };
114
+ }
115
+
116
+ /** The cron's instance-creation duty for one resident (item 7): at most one
117
+ * instance per ten-minute bucket, never while a cycle is live, only once the
118
+ * cadence has elapsed (`shouldCreateRefreshInstance`); the id is
73
119
  * deterministic per resident and bucket, so a second firing in one bucket
74
120
  * meets the engine's duplicate-id refusal, which is the expected no-op. A
75
- * skipped live cycle and a duplicate are recorded on the row for `/status`.
76
- * An `alarm` resident answers null: nothing here touches it. */
121
+ * skipped live cycle and a duplicate are recorded on the row for `/status`. */
77
122
  export async function createRefreshInstance(
78
123
  env: Env,
79
124
  stub: ReturnType<typeof residentStub>,
80
125
  resource: string,
81
126
  row: RefreshRow,
82
- ): Promise<RefreshInstanceAction | null> {
83
- if (row.lifecycle !== "workflow") return null;
127
+ ): Promise<RefreshInstanceAction> {
84
128
  const now = systemClock();
85
129
  const decision = shouldCreateRefreshInstance(row, now, {
86
130
  intervalS: REFRESH_INTERVAL_S,
87
131
  idleIntervalS: IDLE_REFRESH_INTERVAL_S,
88
132
  });
89
- const slug = resource.slice("repo:".length);
90
- const slash = slug.indexOf("/");
91
- const id = refreshInstanceId(slug.slice(0, slash), slug.slice(slash + 1), now);
133
+ const id = bucketInstanceId(resource, now);
92
134
  if (!decision.create) {
93
135
  if (decision.why === "mid-cycle" || decision.why === "running")
94
136
  await stub.recordRefreshSkipped(id, now, decision.why);
95
137
  return { id, action: "skipped", why: decision.why };
96
138
  }
97
- try {
98
- await env.RESIDENT_REFRESH.create({ id, params: { resource } });
99
- } catch (err) {
100
- const message = errMsg(err);
101
- // The engine refuses an id that names an instance still inside its
102
- // retention, and the refusal carries no code — so the id is asked, not
103
- // the wording: an instance that answers for it exists, and the refusal
104
- // was the duplicate it looks like. Any other failure stays a failure.
105
- if (await refreshInstanceExists(env, id)) {
106
- await stub.recordRefreshSkipped(id, now, "duplicate");
107
- return { id, action: "duplicate", why: "duplicate" };
108
- }
109
- console.error(`resident-watchdog: creating refresh instance ${id} failed — ${message}`);
110
- return { id, action: "failed", why: residentText(message) };
111
- }
112
- await stub.recordRefreshInstance(id, now);
113
- console.log(`resident-watchdog: created refresh instance ${id}`);
114
- return { id, action: "created", why: decision.why };
139
+ return createInstance(env, stub, resource, id, now, decision.why);
140
+ }
141
+
142
+ /** The admin `refresh-now` op (item 13): this bucket's instance, created now,
143
+ * due or not — the cycle the cron would create at its next firing, started
144
+ * on demand. Never beside a live cycle (`refreshCycleBlocked`: a running
145
+ * instance, a young `refreshing` marker, a resident that is not serving —
146
+ * answered `skipped` with the why), and under the bucket's deterministic id,
147
+ * so a bucket the cron already served answers `duplicate` naming that
148
+ * instance: the next bucket is at most ten minutes away. */
149
+ export async function createRefreshInstanceNow(
150
+ env: Env,
151
+ stub: ReturnType<typeof residentStub>,
152
+ resource: string,
153
+ ): Promise<RefreshInstanceAction> {
154
+ const now = systemClock();
155
+ const id = bucketInstanceId(resource, now);
156
+ const blocked = refreshCycleBlocked(await stub.refreshRow(), now);
157
+ if (blocked) return { id, action: "skipped", why: blocked };
158
+ return createInstance(env, stub, resource, id, now, "refresh-now");
115
159
  }
116
160
 
117
161
  /** What an instance answers when it ends: small facts for the engine's record. */
@@ -119,33 +163,42 @@ interface RefreshInstanceSummary {
119
163
  instance: string;
120
164
  /** `ok`, or the word a gate or a failure ended the cycle with. */
121
165
  outcome: string;
166
+ /** The step the cycle's verdict came from. */
122
167
  step: "fetch" | "install" | "build" | "snapshot";
123
168
  action?: RefreshPlan["action"];
124
169
  sha?: string;
170
+ /** The housekeeping steps' answers, when they ran. */
171
+ swept?: { evicted: number; kept: number } | null;
172
+ measured?: boolean;
125
173
  }
126
174
 
127
175
  /** One refresh cycle as one short Workflow instance: `fetch`, `install` (only
128
176
  * when the plan moved the lockfile key), `build` (only when the branch
129
- * moved), `snapshot` — each a `step.do` calling the resident's own step
130
- * method through the DO stub, under the retry policy `REFRESH_STEP_RETRIES`
131
- * (six attempts, thirty seconds apart, doubling: about 15.5 minutes, past
132
- * the 3 to 10 minutes a resident Worker rollover takes to settle) and a
133
- * timeout equal to the method's own budget (never above the engine's 30
134
- * minutes). Inputs to a step are the event's `resource`, the instance id and
135
- * previous steps' returns — refs, shas, a key, a path — never a payload and
136
- * never a credential. A step killed from outside (the container replaced
137
- * under it) throws and the engine retries it into the same idempotent
138
- * method; a gate that ends the cycle (idle, a container restart) or a
139
- * failure of the repository's own (recorded as `degraded`, the last snapshot
140
- * still serving) ends the instance with that word, and the next cron firing
141
- * creates the next one from the row's state. The instance runs one cycle
142
- * and returns: it is created by the watchdog cron per resident and
143
- * ten-minute bucket (`createRefreshInstance`), never a loop.
177
+ * moved), `snapshot`, then the housekeeping `sweep` (the worktree
178
+ * inactivity eviction, item 23) and `measure` (the disk sample, item 55) —
179
+ * each a `step.do` calling the resident's own step method through the DO
180
+ * stub, under the retry policy `REFRESH_STEP_RETRIES` (six attempts, thirty
181
+ * seconds apart, doubling: about 15.5 minutes, past the 3 to 10 minutes a
182
+ * resident Worker rollover takes to settle) and a timeout equal to the
183
+ * method's own budget (never above the engine's 30 minutes). Inputs to a
184
+ * step are the event's `resource`, the instance id and previous steps'
185
+ * returns — refs, shas, a key, a path — never a payload and never a
186
+ * credential. A step killed from outside (the container replaced under it)
187
+ * throws and the engine retries it into the same idempotent method; so does
188
+ * a gate that stopped the container on purpose (a stale image, a disk-full
189
+ * recycle) — the retry finds the container back and continues the cycle. A
190
+ * gate that ends the cycle (idle, an offboard) or a failure of the
191
+ * repository's own (recorded as `degraded`, the last snapshot still serving)
192
+ * ends the cycle with that word; the housekeeping steps still run for every
193
+ * verdict but an offboard (nothing is left to keep), and neither wakes a
194
+ * slept container. The instance runs one cycle and returns: it is created by
195
+ * the watchdog cron per resident and ten-minute bucket
196
+ * (`createRefreshInstance`) or by the admin `refresh-now` op, never a loop;
197
+ * the next cron firing creates the next one from the row's state.
144
198
  *
145
199
  * The run is the cycle's root span, `resident.refresh` carrying the instance
146
200
  * id (docs/reference/specs/tracing.md item 25), with every command a step ran
147
- * grafted under it as a `resident.<step>` child — the same shape the alarm's
148
- * root has. */
201
+ * grafted under it as a `resident.<step>` child. */
149
202
  export class ResidentRefresh extends WorkflowEntrypoint<Env, RefreshInstanceParams> {
150
203
  async run(
151
204
  event: Readonly<WorkflowEvent<RefreshInstanceParams>>,
@@ -167,7 +220,7 @@ export class ResidentRefresh extends WorkflowEntrypoint<Env, RefreshInstancePara
167
220
  clipAt: systemClock(),
168
221
  });
169
222
  const retries = REFRESH_STEP_RETRIES;
170
- /** The word a step that did not finish ends the instance with, and its outcome for the root. */
223
+ /** The word a step that did not finish ends the cycle with. */
171
224
  const ended = (
172
225
  at: RefreshInstanceSummary["step"],
173
226
  answer: { status: "stopped"; why: string } | { status: "failed"; reason: string },
@@ -178,53 +231,86 @@ export class ResidentRefresh extends WorkflowEntrypoint<Env, RefreshInstancePara
178
231
  });
179
232
  let summary: RefreshInstanceSummary | undefined;
180
233
  try {
234
+ let cycle: RefreshInstanceSummary | undefined;
181
235
  const fetched = await step.do("fetch", { retries, timeout: stepTimeoutMs(REFRESH_FETCH_STEP_BUDGET_MS) }, () =>
182
236
  stub.refreshInstanceFetch({ resource, instance }),
183
237
  );
184
238
  graft(fetched);
185
- if (fetched.status !== "done") return (summary = ended("fetch", fetched));
186
- let depsEntry: string | null = null;
187
- if (fetched.install) {
188
- const installed = await step.do(
189
- "install",
190
- { retries, timeout: stepTimeoutMs(REFRESH_INSTALL_STEP_BUDGET_MS) },
191
- () => stub.refreshInstanceInstall({ resource, instance, sha: fetched.sha, lockfileKey: fetched.lockfileKey }),
192
- );
193
- graft(installed);
194
- if (installed.status !== "done") return (summary = ended("install", installed));
195
- depsEntry = installed.entry;
239
+ if (fetched.status !== "done") cycle = ended("fetch", fetched);
240
+ else {
241
+ let depsEntry: string | null = null;
242
+ if (fetched.install) {
243
+ const installed = await step.do(
244
+ "install",
245
+ { retries, timeout: stepTimeoutMs(REFRESH_INSTALL_STEP_BUDGET_MS) },
246
+ () =>
247
+ stub.refreshInstanceInstall({ resource, instance, sha: fetched.sha, lockfileKey: fetched.lockfileKey }),
248
+ );
249
+ graft(installed);
250
+ if (installed.status !== "done") cycle = ended("install", installed);
251
+ else depsEntry = installed.entry;
252
+ }
253
+ if (cycle === undefined && fetched.action !== "unchanged") {
254
+ const built = await step.do("build", { retries, timeout: stepTimeoutMs(REFRESH_BUILD_STEP_BUDGET_MS) }, () =>
255
+ stub.refreshInstanceBuild({
256
+ resource,
257
+ instance,
258
+ sha: fetched.sha,
259
+ factsSha: fetched.factsSha,
260
+ lockfileKey: fetched.lockfileKey,
261
+ depsEntry,
262
+ }),
263
+ );
264
+ graft(built);
265
+ if (built.status !== "done") cycle = ended("build", built);
266
+ }
267
+ if (cycle === undefined) {
268
+ const snapped = await step.do(
269
+ "snapshot",
270
+ { retries, timeout: stepTimeoutMs(REFRESH_SNAPSHOT_STEP_BUDGET_MS) },
271
+ () =>
272
+ stub.refreshInstanceSnapshot({
273
+ resource,
274
+ instance,
275
+ ref: fetched.ref,
276
+ sha: fetched.sha,
277
+ lockfileKey: fetched.lockfileKey,
278
+ action: fetched.action,
279
+ mintError: fetched.mintError,
280
+ }),
281
+ );
282
+ graft(snapped);
283
+ cycle =
284
+ snapped.status !== "done"
285
+ ? ended("snapshot", snapped)
286
+ : { instance, outcome: "ok", step: "snapshot", action: fetched.action, sha: fetched.sha };
287
+ }
196
288
  }
197
- if (fetched.action !== "unchanged") {
198
- const built = await step.do("build", { retries, timeout: stepTimeoutMs(REFRESH_BUILD_STEP_BUDGET_MS) }, () =>
199
- stub.refreshInstanceBuild({
200
- resource,
201
- instance,
202
- sha: fetched.sha,
203
- factsSha: fetched.factsSha,
204
- lockfileKey: fetched.lockfileKey,
205
- depsEntry,
206
- }),
289
+ // Housekeeping rides on every instance whatever the cycle's verdict — a
290
+ // parked resident still releases its idle bindings and re-measures, and
291
+ // neither step wakes a slept container — except an offboarded one:
292
+ // nothing is left to keep.
293
+ if (cycle.outcome !== "offboarded") {
294
+ const swept = await step.do("sweep", { retries, timeout: stepTimeoutMs(REFRESH_SWEEP_STEP_BUDGET_MS) }, () =>
295
+ stub.refreshInstanceSweep({ resource, instance }),
207
296
  );
208
- graft(built);
209
- if (built.status !== "done") return (summary = ended("build", built));
297
+ graft(swept);
298
+ const measured = await step.do(
299
+ "measure",
300
+ { retries, timeout: stepTimeoutMs(REFRESH_MEASURE_STEP_BUDGET_MS) },
301
+ () => stub.refreshInstanceMeasure({ instance }),
302
+ );
303
+ graft(measured);
304
+ cycle = {
305
+ ...cycle,
306
+ swept:
307
+ swept.status === "done" && swept.result
308
+ ? { evicted: swept.result.evicted.length, kept: swept.result.kept }
309
+ : null,
310
+ measured: measured.status === "done" && measured.result !== null && measured.result.measured,
311
+ };
210
312
  }
211
- const snapped = await step.do(
212
- "snapshot",
213
- { retries, timeout: stepTimeoutMs(REFRESH_SNAPSHOT_STEP_BUDGET_MS) },
214
- () =>
215
- stub.refreshInstanceSnapshot({
216
- resource,
217
- instance,
218
- ref: fetched.ref,
219
- sha: fetched.sha,
220
- lockfileKey: fetched.lockfileKey,
221
- action: fetched.action,
222
- mintError: fetched.mintError,
223
- }),
224
- );
225
- graft(snapped);
226
- if (snapped.status !== "done") return (summary = ended("snapshot", snapped));
227
- return (summary = { instance, outcome: "ok", step: "snapshot", action: fetched.action, sha: fetched.sha });
313
+ return (summary = cycle);
228
314
  } catch (err) {
229
315
  // A step out of retries: the engine records the failed instance by id;
230
316
  // the row keeps its last state and the next cron firing starts the next cycle.
@@ -19,20 +19,25 @@ import type { Env } from "./worker";
19
19
  * resident is re-warmed before the platform can sleep it. Bump together. */
20
20
  export const SLEEP_AFTER = "20m";
21
21
 
22
- /** Refresh alarm cadence (seconds). Each resident DO self-reschedules this
23
- * alarm (per-resident alarms own freshness; the sparse cron is only the
24
- * watchdog); it doubles as the keep-warm heartbeat, so it must stay below
25
- * SLEEP_AFTER. Matches the watchdog cron so a killed chain is re-armed
26
- * within one refresh interval. */
22
+ /** The refresh cadence (seconds): one instance per resident per ten-minute
23
+ * bucket, created by the watchdog cron, so the cron's own period is the
24
+ * cadence. The cycle doubles as the keep-warm heartbeat (its steps call into
25
+ * the container), so it must stay below SLEEP_AFTER. */
27
26
  export const REFRESH_INTERVAL_S = 600;
28
27
 
29
- /** How far out an idle resident's cycle is parked (seconds): the alarm's idle
30
- * re-arm (`IDLE_AFTER_S` in worker.ts decides idleness) and the cron's idle
31
- * cadence, in whole ten-minute buckets since the last instance. */
28
+ /** How far out an idle resident's next cycle is (seconds): the cron's idle
29
+ * cadence (`IDLE_AFTER_S` in worker.ts decides idleness), in whole
30
+ * ten-minute buckets since the last instance. */
32
31
  export const IDLE_REFRESH_INTERVAL_S = 6 * 60 * 60;
33
32
 
34
- /** Exec budgets. The DO alarm handler has a ~15-minute platform wall clock;
35
- * every schedule callback's step budgets are chosen to fit under it. */
33
+ /** The thread user pool the image carries (`worker2`..`worker17`): one OS
34
+ * user per attached thread, and the bound on how many bindings one sweep
35
+ * step can have to check. */
36
+ export const THREAD_POOL_SIZE = 16;
37
+
38
+ /** Exec budgets. Each is a refresh step's own budget too (refresh.ts), so a
39
+ * step timeout and the command timeout it bounds agree; every one is under
40
+ * the engine's 30-minute step ceiling. */
36
41
  export const DEFAULT_EXEC_TIMEOUT_MS = 60_000;
37
42
  export const GIT_NETWORK_TIMEOUT_MS = 5 * 60_000;
38
43
  export const REFRESH_BUILD_TIMEOUT_MS = 5 * 60_000;