@coreplane/switchboard 1.260.0 → 1.260.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/bin/switchboard.js +15 -11
  2. package/dist/assets/config/config.example.yaml +1 -0
  3. package/dist/assets/deploy/cloudflare/worker.ts +17 -56
  4. package/dist/assets/deploy/cloudflare/wrangler.template.jsonc +6 -0
  5. package/dist/assets/deploy/cloudflare-memory/backgroundTasks.ts +25 -0
  6. package/dist/assets/deploy/cloudflare-memory/worker.ts +36 -16
  7. package/dist/assets/deploy/cloudflare-resident/worker.ts +39 -18
  8. package/dist/assets/deploy/profile.example.json +2 -1
  9. package/dist/assets/package-lock.json +3 -3
  10. package/dist/assets/package.json +1 -1
  11. package/dist/assets/source.json +3 -3
  12. package/dist/assets/src/agents/registry.ts +2 -2
  13. package/dist/assets/src/core/budgets.ts +19 -6
  14. package/dist/assets/src/core/coordinator/driver.ts +47 -9
  15. package/dist/assets/src/core/pipelineStanding.ts +2 -0
  16. package/dist/assets/src/core/reviewedHead.ts +16 -2
  17. package/dist/assets/src/core/runEvents.ts +9 -1
  18. package/dist/assets/src/core/runLedger/types.ts +3 -3
  19. package/dist/assets/src/core/runRecord.ts +28 -8
  20. package/dist/assets/src/core/ship/coordinator.ts +248 -44
  21. package/dist/assets/src/core/ship/renewal.ts +2 -0
  22. package/dist/assets/src/deploy/liveGate.ts +8 -0
  23. package/dist/assets/src/deploy/profile.ts +8 -0
  24. package/dist/assets/src/deploy/restart.ts +49 -35
  25. package/dist/assets/web/dist/.vite/manifest.json +67 -67
  26. package/dist/assets/web/dist/assets/{DeliveryPage-DvMrWUg7.js → DeliveryPage-62qCSLvk.js} +1 -1
  27. package/dist/assets/web/dist/assets/{HomePage-CcGEJ4w0.js → HomePage-Dk3HRBc3.js} +1 -1
  28. package/dist/assets/web/dist/assets/{PendingTurnRow-BzGVxDYs.js → PendingTurnRow-CeTGOXUy.js} +1 -1
  29. package/dist/assets/web/dist/assets/{PlanePage-Arj9cyd5.js → PlanePage-RD0M6O6W.js} +1 -1
  30. package/dist/assets/web/dist/assets/{ResidentDetailPage-CBPPe4Sj.js → ResidentDetailPage-Dl2KId95.js} +1 -1
  31. package/dist/assets/web/dist/assets/{ResidentsIndexPage-CEpXgGo7.js → ResidentsIndexPage-DBy43I1R.js} +1 -1
  32. package/dist/assets/web/dist/assets/{RunFoldRow-BJq7tnLr.js → RunFoldRow-CwHDL-Mo.js} +1 -1
  33. package/dist/assets/web/dist/assets/{RunRoutePage-DWiny7LN.js → RunRoutePage-Du1HqLZs.js} +3 -3
  34. package/dist/assets/web/dist/assets/{RunsIndexPage-Cxufw5sV.js → RunsIndexPage-BZphy3Hr.js} +1 -1
  35. package/dist/assets/web/dist/assets/{ScheduledPage-SBcuDdVg.js → ScheduledPage-B7jh898t.js} +1 -1
  36. package/dist/assets/web/dist/assets/{SettingsPage-DrD_vWdp.js → SettingsPage-3uLEsiiE.js} +1 -1
  37. package/dist/assets/web/dist/assets/{SilentTurn-Bjl7EPEN.js → SilentTurn-CKxSxnQC.js} +1 -1
  38. package/dist/assets/web/dist/assets/{StatusDot-BqteQTav.js → StatusDot-x7vcK0iE.js} +1 -1
  39. package/dist/assets/web/dist/assets/{Tooltip-KRqJXEPN.js → Tooltip-wVXCWhMp.js} +1 -1
  40. package/dist/assets/web/dist/assets/{UnitRoutePage-Cv28FV4I.js → UnitRoutePage-V5BXK7FH.js} +1 -1
  41. package/dist/assets/web/dist/assets/budgets-KNNkT4PZ.js +1 -0
  42. package/dist/assets/web/dist/assets/{dist-CTASUSno.js → dist-C-XGnvM4.js} +1 -1
  43. package/dist/assets/web/dist/assets/{indexRow-B1YBy_IL.js → indexRow-Dxg7C8s9.js} +1 -1
  44. package/dist/assets/web/dist/assets/{main-eTAe9-hk.js → main-DAL--hyj.js} +2 -2
  45. package/dist/assets/web/dist/assets/sseReplay-BqgOggyQ.js +11 -0
  46. package/dist/cli.js +1405 -436
  47. package/package.json +1 -1
  48. package/dist/assets/web/dist/assets/budgets-DQXVllsm.js +0 -1
  49. package/dist/assets/web/dist/assets/sseReplay-BcsNbn2j.js +0 -11
@@ -1,25 +1,29 @@
1
1
  #!/usr/bin/env node
2
- // The `switchboard` bin: this file hands the process to dist/cli.js, the bundle
3
- // the build writes (build.mts; docs/reference/specs/packaging.md item 1). A
4
- // committed entry rather than the bundle itself, because npm links a bin only
5
- // when its target exists at install time, and the gitignored dist/ does not
6
- // until the build runs: with dist/cli.js as the bin, `npx <the package>` inside
7
- // the checkout — where npx prefers the workspace over the registry — found no
8
- // link and died with `sh: switchboard: command not found`. The published package
9
- // always carries the bundle, so there this is one extra module load; a checkout
10
- // that has not built the package is told what to run.
2
+ // The published `switchboard` bin hands the process to the bundle beside it.
3
+ // In a repository checkout npm resolves the package name to this workspace too,
4
+ // but an ignored dist/ may be older than source. The checkout therefore never
5
+ // trusts dist: operators run the root source script, while a tarball runs only
6
+ // the bundle that its build packed.
11
7
  import { existsSync } from "node:fs";
12
8
  import { fileURLToPath } from "node:url";
13
9
 
14
10
  const bundle = new URL("../dist/cli.js", import.meta.url);
11
+ const checkoutRoot = new URL("../../../", import.meta.url);
12
+ const runsFromCheckout =
13
+ existsSync(new URL("project.json", checkoutRoot)) && existsSync(new URL("src/cli.ts", checkoutRoot));
15
14
 
16
- if (!existsSync(bundle)) {
15
+ if (runsFromCheckout) {
17
16
  console.error(
18
- `switchboard: ${fileURLToPath(bundle)} is missing — the package is not built. Inside the checkout, npx runs this workspace, not the published package: run \`npm run build -w packages/switchboard\` first, or use the checkout's CLI, \`npm run cli -- <group> <verb> …\`.`,
17
+ "switchboard: npx resolved the package name to this checkout; its ignored dist/ is not authoritative. Run `npm run --silent cli -- <group> <verb> …` from the checkout root instead.",
19
18
  );
20
19
  process.exit(1);
21
20
  }
22
21
 
22
+ if (!existsSync(bundle)) {
23
+ console.error(`switchboard: ${fileURLToPath(bundle)} is missing — the published package is incomplete.`);
24
+ process.exit(1);
25
+ }
26
+
23
27
  // From here the bundle is the script: its entry claim (src/invokedAsScript.ts)
24
28
  // and its usage spelling (`programName`, src/cli.ts) read argv[1].
25
29
  process.argv[1] = fileURLToPath(bundle);
@@ -172,6 +172,7 @@ defaults:
172
172
  # "config set channel ..." are stored in data/overrides.json and win over these.
173
173
  # channels:
174
174
  # slack:C012345:
175
+ # repo: acme/api # default repository for repo-bound asks in this channel
175
176
  # agent: review
176
177
  # models:
177
178
  # review: anthropic/claude-opus-5
@@ -28,14 +28,10 @@ import {
28
28
  import {
29
29
  authenticateIngressBearer,
30
30
  authenticateRestart,
31
+ authorizeRestartDeployer,
31
32
  decideRestart,
32
- parseRestartAuthorization,
33
33
  parseRestartRequest,
34
- RESTART_AUTHORIZE_PATH,
35
- RESTART_SUBJECT_HEADER,
36
34
  restartResponse,
37
- stripRestartSubject,
38
- type RestartAuth,
39
35
  type RestartOutcome,
40
36
  } from "../../src/deploy/restart.ts";
41
37
  import {
@@ -89,7 +85,8 @@ export interface Env {
89
85
  ACCESS_TEAM_DOMAIN?: string; // live-view SSO gate: Cloudflare Access team domain (JWKS + iss)
90
86
  ACCESS_AUD?: string; // live-view SSO gate: Cloudflare Access application AUD tag
91
87
  DASHBOARD_TOKEN?: string; // dashboard auth `token` strategy: the bearer (the default env name; config may name another)
92
- SWITCHBOARD_INGRESS_TOKENS?: string; // enables HTTP /ingress + MCP /mcp (JSON token→identity map); the `cron` entry is what scheduled runs present; an entry whose `http:<subject>` actor holds `deploy:write` in the bot's config may POST /admin/restart
88
+ SWITCHBOARD_INGRESS_TOKENS?: string; // enables HTTP /ingress + MCP /mcp (JSON token→identity map); the `cron` entry is what scheduled runs present
89
+ SWITCHBOARD_RESTART_DEPLOYER?: string; // deployment-profile grant: the token-map subject allowed to POST /admin/restart; Worker-only, never runtime config
93
90
  BRAVE_SEARCH_API_KEY?: string; // web_search backend (Brave); web_fetch works without it
94
91
  GITHUB_WEBHOOK_SECRET?: string; // check-run intake: signs POST /webhooks/github; absent, the intake answers 503 disabled
95
92
  CF_ANALYTICS_TOKEN?: string; // costs dash: Cloudflare API token, Account Analytics:Read only
@@ -211,37 +208,6 @@ export class SwitchboardServer extends Container<Env> {
211
208
  return this.restartRunning(opts);
212
209
  }
213
210
 
214
- /**
215
- * `deploy restart`, authorized: the Worker knows WHO the bearer is (the token
216
- * map); WHETHER that identity may restart is the bot's config (`grants` —
217
- * authorization.md item 9), which only the container holds. So ask it —
218
- * `POST /admin/restart/authorize` with the authenticated subject in
219
- * `RESTART_SUBJECT_HEADER` (never the bearer: the container may still hold the
220
- * token map from before a rotation) — and stop only on a 200. A container that
221
- * is not running is started first (the bot must answer): if the bearer is
222
- * allowed, that start already put the current env live, so nothing is
223
- * stopped and the outcome is `not-running`, exactly as before; if not, the
224
- * refusal is relayed and the started container simply keeps serving.
225
- */
226
- async restartAuthorized(
227
- subject: string,
228
- opts: { force: boolean },
229
- ): Promise<{ auth: RestartAuth; outcome?: RestartOutcome }> {
230
- const wasRunning = this.ctx.container?.running === true;
231
- await this.startBot();
232
- const answer = await this.containerFetch(
233
- new Request(`${INTERNAL}${RESTART_AUTHORIZE_PATH}`, {
234
- method: "POST",
235
- headers: { [RESTART_SUBJECT_HEADER]: subject },
236
- }),
237
- this.defaultPort,
238
- );
239
- const auth = parseRestartAuthorization(answer.status, await answer.text().catch(() => ""));
240
- if (!auth.ok) return { auth };
241
- if (!wasRunning) return { auth, outcome: { kind: "not-running" } };
242
- return { auth, outcome: await this.restartRunning(opts) };
243
- }
244
-
245
211
  /** The stop itself, for a running container: the preflight's refusal rules over `/healthz`, then SIGTERM. */
246
212
  private async restartRunning(opts: { force: boolean }): Promise<RestartOutcome> {
247
213
  const health = await this.containerFetch(new Request(`${INTERNAL}/healthz`), this.defaultPort);
@@ -265,12 +231,12 @@ export class SwitchboardServer extends Container<Env> {
265
231
  }
266
232
  }
267
233
 
268
- /** `POST /admin/restart` — the operator surface behind `deploy restart`
269
- * (src/deploy/restart.ts documents the authorization choice: a
270
- * SWITCHBOARD_INGRESS_TOKENS bearer whose `http:<subject>` actor holds
271
- * `deploy:write` in the bot's config). The Worker authenticates the bearer
272
- * against the map it holds — an unknown bearer never touches the container —
273
- * and the Container DO asks the bot for the grant before stopping anything.
234
+ /** `POST /admin/restart` — the operator surface behind `deploy restart`.
235
+ * The Worker authenticates the bearer against its token map, then authorizes
236
+ * that subject against SWITCHBOARD_RESTART_DEPLOYER, rendered from the
237
+ * deployment profile (or set directly as a Worker var). Runtime config is
238
+ * never consulted: this route must remain usable when that document is what
239
+ * the restart is repairing.
274
240
  * Body `{ "force": true }` bypasses the fail-closed refusals (no JSON body,
275
241
  * impossible `inFlight`); runs in flight or a drain warn and never refuse. */
276
242
  async function handleAdminRestart(request: Request, env: Env): Promise<Response> {
@@ -283,14 +249,16 @@ async function handleAdminRestart(request: Request, env: Env): Promise<Response>
283
249
  console.warn(`[restart] ${authn.status} — ${authn.reason}`);
284
250
  return json(authn.status, { ok: false, error: authn.reason });
285
251
  }
252
+ const auth = authorizeRestartDeployer(authn.identity.subject, env.SWITCHBOARD_RESTART_DEPLOYER);
253
+ if (!auth.ok) {
254
+ console.warn(`[restart] ${auth.status} — ${auth.reason}`);
255
+ return json(auth.status, { ok: false, error: auth.reason });
256
+ }
286
257
  const parsed = parseRestartRequest(await request.text().catch(() => ""));
287
258
  if (!parsed.ok) return json(400, { ok: false, error: parsed.reason });
288
- let auth: RestartAuth;
289
- let outcome: RestartOutcome | undefined;
259
+ let outcome: RestartOutcome;
290
260
  try {
291
- ({ auth, outcome } = await getContainer(env.SWITCHBOARD, INSTANCE).restartAuthorized(authn.identity.subject, {
292
- force: parsed.force,
293
- }));
261
+ outcome = await getContainer(env.SWITCHBOARD, INSTANCE).restart({ force: parsed.force });
294
262
  } catch (err) {
295
263
  // The container's /healthz probe or the DO call threw (container mid-transition,
296
264
  // port not answering): fail closed in the route's own JSON shape so the CLI reads
@@ -299,11 +267,6 @@ async function handleAdminRestart(request: Request, env: Env): Promise<Response>
299
267
  console.error(`[restart] ${authn.identity.subject} → error before stop: ${reason}`);
300
268
  return json(500, { ok: false, error: `restart failed before stopping anything: ${reason}` });
301
269
  }
302
- if (!auth.ok || outcome === undefined) {
303
- const refusal = auth.ok ? { status: 503 as const, reason: "restart disabled: no outcome" } : auth;
304
- console.warn(`[restart] ${refusal.status} — ${refusal.reason}`);
305
- return json(refusal.status, { ok: false, error: refusal.reason });
306
- }
307
270
  console.log(
308
271
  `[restart] ${auth.subject} → ${outcome.kind}${outcome.kind === "refused" ? `: ${outcome.problems.join("; ")}` : ""}`,
309
272
  );
@@ -494,9 +457,7 @@ export default {
494
457
  // caller sent is stripped, and what the container sees carries this
495
458
  // Worker's own root. A static asset or the live view's SSE stream gets no
496
459
  // root; a refusal's line is dropped by the sink's filter.
497
- // The restart-subject header is the Worker's own word to the container (the
498
- // authorize call below); a caller cannot be allowed to speak it.
499
- const inbound = stripRestartSubject(stripTraceContext(request));
460
+ const inbound = stripTraceContext(request);
500
461
  const route = shimRoute(pathname);
501
462
  if (route === undefined) return withLength(await getContainer(env.SWITCHBOARD, INSTANCE).fetch(inbound));
502
463
  const root = tracer.start("bot-shim.fetch", { sinks: traceSinks, attrs: { route } });
@@ -31,6 +31,12 @@
31
31
  "ACCESS_TEAM_DOMAIN": "{{access.teamDomain}}",
32
32
  "ACCESS_AUD": "{{access.aud}}",
33
33
  // {{/if}}
34
+ // The restart route's deployment-side grant. The Worker authenticates a
35
+ // bearer through SWITCHBOARD_INGRESS_TOKENS and compares its subject here;
36
+ // it never asks the runtime config it may be restarting to repair.
37
+ // {{#if restart}}
38
+ "SWITCHBOARD_RESTART_DEPLOYER": "{{restart.deployer}}",
39
+ // {{/if}}
34
40
  // The state Worker (deploy/cloudflare-memory/): where worker.ts records each
35
41
  // scheduled firing for the /runs Scheduled panel, with MEMORY_TOKEN.
36
42
  // A profile without a memory Worker renders no var: firings go unrecorded.
@@ -0,0 +1,25 @@
1
+ type WaitUntilContext = Pick<DurableObjectState, "waitUntil">;
2
+
3
+ const TASKS = Symbol.for("switchboard.memory-worker.pending-background-tasks");
4
+ type PendingTasks = Map<Promise<unknown>, string>;
5
+
6
+ const pendingTasks = (): PendingTasks => {
7
+ const root = globalThis as typeof globalThis & { [TASKS]?: PendingTasks };
8
+ return (root[TASKS] ??= new Map());
9
+ };
10
+
11
+ /** Keep deliberately detached Worker work in the runtime's request lifetime.
12
+ * The process-wide label set is also the test seam: a case cannot finish while
13
+ * work it started is still able to contend with the next case. */
14
+ export function holdBackgroundTask(context: WaitUntilContext, label: string, task: Promise<unknown>): void {
15
+ const pending = pendingTasks();
16
+ const held = task.finally(() => pending.delete(held));
17
+ pending.set(held, label);
18
+ context.waitUntil(held);
19
+ }
20
+
21
+ export function assertNoPendingBackgroundTasks(): void {
22
+ const labels = [...pendingTasks().values()];
23
+ if (labels.length === 0) return;
24
+ throw new Error(`test ended with ${labels.length} pending background task(s): ${labels.join(", ")}`);
25
+ }
@@ -78,6 +78,7 @@ import {
78
78
  selectReclaim,
79
79
  } from "../../src/core/runLedger/decisions.ts";
80
80
  import { intakeReceiptRetentionMs, minutesToMs, PLANE } from "../../src/core/budgets.ts";
81
+ import { holdBackgroundTask } from "./backgroundTasks.ts";
81
82
  import {
82
83
  causeOfClose,
83
84
  causeOfReclaim,
@@ -2058,7 +2059,7 @@ export class RunHistoryDO extends DurableObject<Env> {
2058
2059
  /** The admission ask (`POST /plane/admit`, record 0064 "The queue"): one
2059
2060
  * transaction decides and writes — `admitted` reserves the thread,
2060
2061
  * `queued` stores the request under the minted id. */
2061
- planeAdmit(
2062
+ async planeAdmit(
2062
2063
  post: {
2063
2064
  runId: string;
2064
2065
  requester: string;
@@ -2070,7 +2071,7 @@ export class RunHistoryDO extends DurableObject<Env> {
2070
2071
  reaskMs?: number;
2071
2072
  },
2072
2073
  now: number,
2073
- ): PlaneAskAnswer {
2074
+ ): Promise<PlaneAskAnswer> {
2074
2075
  let answer!: PlaneAskAnswer;
2075
2076
  this.ctx.storage.transactionSync(() => {
2076
2077
  if (post.reaskMs !== undefined)
@@ -2089,7 +2090,7 @@ export class RunHistoryDO extends DurableObject<Env> {
2089
2090
  this.applyPlaneWrites(decision.writes);
2090
2091
  answer = planeAskAnswerOf(decision, post.runId);
2091
2092
  });
2092
- if (answer.kind === "queued") void this.ensurePlaneReaskAlarm(now);
2093
+ if (answer.kind === "queued") await this.settlePlaneReaskAlarmAfterCommit(now);
2093
2094
  console.log(
2094
2095
  `[plane/admit] ${post.threadKey} → ${answer.kind}${answer.kind === "queued" ? ` position ${answer.position}` : ""} (run ${post.runId})`,
2095
2096
  );
@@ -2152,7 +2153,10 @@ export class RunHistoryDO extends DurableObject<Env> {
2152
2153
 
2153
2154
  /** A refusal-by-name the bot met at attach or exec (`POST /plane/observe`,
2154
2155
  * record 0064): an admitted run re-enters the queue at its old position. */
2155
- planeObserve(post: { runId: string; resident: string; refusal: string }, now: number): { reentered: boolean } {
2156
+ async planeObserve(
2157
+ post: { runId: string; resident: string; refusal: string },
2158
+ now: number,
2159
+ ): Promise<{ reentered: boolean }> {
2156
2160
  let reentered = false;
2157
2161
  this.ctx.storage.transactionSync(() => {
2158
2162
  const decision = decide(this.planeState(), {
@@ -2165,7 +2169,7 @@ export class RunHistoryDO extends DurableObject<Env> {
2165
2169
  this.applyPlaneWrites(decision.writes);
2166
2170
  reentered = decision.writes.length > 0;
2167
2171
  });
2168
- if (reentered) void this.ensurePlaneReaskAlarm(now);
2172
+ if (reentered) await this.settlePlaneReaskAlarmAfterCommit(now);
2169
2173
  console.log(
2170
2174
  `[plane/observe] run ${post.runId} on ${post.resident}: ${post.refusal.slice(0, 60)} — ${reentered ? "re-entered" : "no-op"}`,
2171
2175
  );
@@ -2194,6 +2198,18 @@ export class RunHistoryDO extends DurableObject<Env> {
2194
2198
  if (set === null || set > due) await this.ctx.storage.setAlarm(due);
2195
2199
  }
2196
2200
 
2201
+ /** Settle alarm I/O without contradicting the plane transaction that already
2202
+ * committed. A scheduling failure may delay a re-ask until another wake,
2203
+ * but returning an error would make the caller proceed or retry while the
2204
+ * durable queued row remains eligible for admission. */
2205
+ private async settlePlaneReaskAlarmAfterCommit(now: number): Promise<void> {
2206
+ try {
2207
+ await this.ensurePlaneReaskAlarm(now);
2208
+ } catch (err) {
2209
+ console.error("[plane/alarm] re-ask scheduling failed after the plane state committed", err);
2210
+ }
2211
+ }
2212
+
2197
2213
  /** The earliest instant the plane must wake at (record 0064): the
2198
2214
  * earliest hosting deadline, lease end or re-ask across its rows. A due
2199
2215
  * already past re-offers at the re-ask cadence, never in a hot loop, and
@@ -2372,21 +2388,24 @@ export class RunHistoryDO extends DurableObject<Env> {
2372
2388
 
2373
2389
  /** The transport's push (record 0064, "Where it lives"): committed effects
2374
2390
  * are pushed to the bot Worker over the service binding, which forwards to
2375
- * the container. Fire and forget — a push that fails is not retried by a
2376
- * timer; the effect rides the next heartbeat or reclaim-sweep answer. */
2391
+ * the container. The response does not wait for this best-effort push, but
2392
+ * the actor does: waitUntil keeps its I/O inside this request's lifetime so
2393
+ * it cannot contend with an unrelated request after the caller moves on. */
2377
2394
  private pushPlaneEffects(effects: PlaneEffect[]): void {
2378
2395
  if (effects.length === 0) return;
2379
2396
  const bot = this.env.BOT;
2380
2397
  if (!bot) return;
2381
- void bot
2382
- .fetch("https://bot/plane/effects", {
2383
- method: "POST",
2384
- headers: {
2385
- "content-type": "application/json",
2386
- authorization: `Bearer ${this.env.MEMORY_TOKEN ?? ""}`,
2387
- },
2388
- body: JSON.stringify({ effects }),
2389
- })
2398
+ const delivery = Promise.resolve()
2399
+ .then(() =>
2400
+ bot.fetch("https://bot/plane/effects", {
2401
+ method: "POST",
2402
+ headers: {
2403
+ "content-type": "application/json",
2404
+ authorization: `Bearer ${this.env.MEMORY_TOKEN ?? ""}`,
2405
+ },
2406
+ body: JSON.stringify({ effects }),
2407
+ }),
2408
+ )
2390
2409
  .then((r) => {
2391
2410
  if (!r.ok)
2392
2411
  console.warn(
@@ -2398,6 +2417,7 @@ export class RunHistoryDO extends DurableObject<Env> {
2398
2417
  `[plane/push] ${effects.length} effect(s) not delivered: ${err instanceof Error ? err.message : String(err)} — they ride the next heartbeat`,
2399
2418
  );
2400
2419
  });
2420
+ holdBackgroundTask(this.ctx, `plane effect push (${effects.map((effect) => effect.id).join(", ")})`, delivery);
2401
2421
  }
2402
2422
 
2403
2423
  /** The unacknowledged effects, oldest first, at most `PLANE_EFFECTS_PER_ANSWER`
@@ -5027,11 +5027,32 @@ export class ResidentDO extends Sandbox<Env> {
5027
5027
  * deploy's alone. Answers `restarted` when a stop was issued, else why
5028
5028
  * not. */
5029
5029
  private async reconcileImage(where: "attach" | "refresh" | "deploy", force = false): Promise<ImageReconcileResult> {
5030
- if (!(await this.isRuntimeActive().catch(() => false))) {
5031
- // Inactivity proves only that the old process is gone, not that its
5032
- // replacement started successfully. Keep the marker: the next refresh's
5033
- // hydration starts the deployed image and reports only after it reaches
5034
- // `warm` (issue 2101).
5030
+ const active = await this.isRuntimeActive().catch((err) => {
5031
+ console.log(`image-reconcile (${where}): activity probe failed — preserving the pending report: ${errMsg(err)}`);
5032
+ return null;
5033
+ });
5034
+ // A failed probe proves neither state. Preserve the marker and its drain
5035
+ // hold until a later reconcile confirms inactivity or hydrates a fresh
5036
+ // active container. An attach with that marker must fail closed even after
5037
+ // the drain backstop reopens: admitting it could wake a pre-deploy image.
5038
+ if (active === null) {
5039
+ if (
5040
+ where === "attach" &&
5041
+ (await this.ctx.storage.get<{ resource: string }>(IMAGE_REPORT_PENDING_KEY)) !== undefined
5042
+ ) {
5043
+ console.log(
5044
+ "image-stale (attach): activity could not be verified while a new-image report is pending — the attach is refused",
5045
+ );
5046
+ return "stale";
5047
+ }
5048
+ return "deferred";
5049
+ }
5050
+ if (!active) {
5051
+ // An inactive container's next start is on the deployed image by
5052
+ // construction. Report on every reconcile: the deploy may have marked
5053
+ // it pending after an earlier idle cycle, and the drain itself refuses
5054
+ // the attach that would otherwise wake it and provide another reporter.
5055
+ await this.reportPendingImageCurrent(where);
5035
5056
  return "inactive";
5036
5057
  }
5037
5058
  const last = THREAD_USERS[THREAD_USERS.length - 1];
@@ -5091,12 +5112,13 @@ export class ResidentDO extends Sandbox<Env> {
5091
5112
  /** The report a held drain waits for (issue 1931): when the deploy's
5092
5113
  * reconcile could not verify this resident's fresh container on the new
5093
5114
  * image, a marker stays in storage and the registry holds the drain. The
5094
- * fact the report stands on is the replacement's own completed hydration:
5095
- * the fresh container has restored its snapshot and reached `warm`. A stop
5096
- * or inactive runtime does not report; each keeps the marker for that next
5097
- * start (issue 2101). The marker is deleted only after a successful report:
5098
- * a transient failure keeps it for the next reconcile, and the drain's
5099
- * `until` is the backstop for a report that never lands. */
5115
+ * fact the report stands on is either an inactive runtime, whose next start
5116
+ * uses the deployed image by construction, or the active replacement's own
5117
+ * completed hydration: the fresh container has restored its snapshot and
5118
+ * reached `warm`. A stop alone does not report (issue 2101). The marker is
5119
+ * deleted only after a successful report: a transient failure keeps it for
5120
+ * the next reconcile, and the drain's `until` is the backstop for a report
5121
+ * that never lands. */
5100
5122
  private async reportPendingImageCurrent(where: string): Promise<boolean> {
5101
5123
  const pending = await this.ctx.storage.get<{ resource: string }>(IMAGE_REPORT_PENDING_KEY);
5102
5124
  if (pending === undefined) return true;
@@ -5125,9 +5147,9 @@ export class ResidentDO extends Sandbox<Env> {
5125
5147
  * an active, quiet container is always cycled, but that stop is not a report.
5126
5148
  * The marker survives until the replacement hydrates and reaches `warm` on
5127
5149
  * the deploy's image (issue 2101: the old probe raced that restore and lost).
5128
- * `verified` true is the fact the reopen may stand on; a cycle or inactive
5129
- * result leaves a hold on the drain, cleared by this resident's later
5130
- * `reportPendingImageCurrent`, never by a timer.
5150
+ * `verified` true is the fact the reopen may stand on: inactivity reports
5151
+ * at once, while an active container's cycle leaves a hold until its fresh
5152
+ * hydration calls `reportPendingImageCurrent`, never until a timer.
5131
5153
  *
5132
5154
  * The hold lands BEFORE the marker: the marker is what lets any concurrent
5133
5155
  * attach/refresh reconcile report, and a report that reaches the registry
@@ -5139,10 +5161,9 @@ export class ResidentDO extends Sandbox<Env> {
5139
5161
  await this.ctx.storage.put(IMAGE_REPORT_PENDING_KEY, { resource });
5140
5162
  const result = await this.reconcileImage("deploy", true);
5141
5163
  if (result === "deferred") return { result, verified: false };
5142
- // Neither inactivity nor a successful stop is verification: only the
5143
- // replacement's completed hydration clears the marker. A concurrent fresh
5144
- // start may have done so while this reconcile yielded, hence the storage
5145
- // read rather than an unconditional false.
5164
+ // Inactivity reports synchronously; a successful stop does not. A
5165
+ // concurrent fresh start may also have hydrated while this reconcile
5166
+ // yielded, hence the storage read rather than a verdict from `result`.
5146
5167
  const verified = (await this.ctx.storage.get(IMAGE_REPORT_PENDING_KEY)) === undefined;
5147
5168
  return { result, verified };
5148
5169
  }
@@ -1,5 +1,5 @@
1
1
  {
2
- "$comment": "The deployment profile: where THIS installation runs. Copy to deploy/profile.json and fill it in (or run `switchboard deploy init`). The account is your Cloudflare account id; every hostname must be under `zone`, a zone in that account, unless the Worker names its own `zone` (also in the account); `workers.bot` is the one required Worker — leave `memory`, `resident` or `sandbox` out and `deploy plan` has no step for them (a bot-only profile is a one-step plan; without `memory` the config is not pushed anywhere). The project's docs site is not a Worker of an installation: it is the project's website, deployed by the project's own CI from project.json, so there is no `docs` entry. `configSource` is where `deploy all` reads the bot's runtime config from before building the image — a path, `github://owner/repo/path@ref` (needs CONFIG_REPO_TOKEN), or `op://Vault/Item/field` (needs OP_SERVICE_ACCOUNT_TOKEN); `secretsSource` is where `secrets put` reads values from — a directory of <NAME> files (the default when absent) or `op://Vault/Item`. `images` is where the bot, resident and sandbox container images come from: `registry` — the release's published images, copied once per version into your account registry by `deploy all` itself (or `deploy images` ahead of it) over HTTPS and referenced from there — no Docker anywhere, a Cloudflare API token with Containers Edit in CLOUDFLARE_API_TOKEN for the copy (an installation deploying published images; what `init` writes from the published package); or `build` (the default when absent) — each Worker's Dockerfile, built by wrangler where `deploy all` runs (a checkout; the project's own production). `artifacts` (optional) names the R2 bucket a run's files move through (docs/reference/specs/execution.md item 20): with it the bot Worker binds the bucket and `deploy` creates it before the upload (the credential then needs Workers R2 Storage: Edit); the bot's runtime config `artifacts.r2.bucket` must say the same name. `metrics` (optional) names the Analytics Engine dataset every finished run's point is written to (docs/reference/specs/run-metrics.md): the state Worker's template then binds it as `RUN_METRICS` with the name beside it in `RUN_METRICS_DATASET` — nothing is created ahead of the deploy, the platform creates the dataset on first write — and the bot's runtime config `metrics.dataset` must say the same name (the bot warns at boot when the two differ). `deploy plan` reads this example when no profile exists; `deploy all` refuses it.",
2
+ "$comment": "The deployment profile: where THIS installation runs. Copy to deploy/profile.json and fill it in (or run `switchboard deploy init`). The account is your Cloudflare account id; every hostname must be under `zone`, a zone in that account, unless the Worker names its own `zone` (also in the account); `workers.bot` is the one required Worker — leave `memory`, `resident` or `sandbox` out and `deploy plan` has no step for them (a bot-only profile is a one-step plan; without `memory` the config is not pushed anywhere). The project's docs site is not a Worker of an installation: it is the project's website, deployed by the project's own CI from project.json, so there is no `docs` entry. `configSource` is where `deploy all` reads the bot's runtime config from before building the image — a path, `github://owner/repo/path@ref` (needs CONFIG_REPO_TOKEN), or `op://Vault/Item/field` (needs OP_SERVICE_ACCOUNT_TOKEN); `restart.deployer` is the SWITCHBOARD_INGRESS_TOKENS subject the Worker itself allows to restart the bot, independent of runtime config so config recovery cannot lock itself out; `secretsSource` is where `secrets put` reads values from — a directory of <NAME> files (the default when absent) or `op://Vault/Item`. `images` is where the bot, resident and sandbox container images come from: `registry` — the release's published images, copied once per version into your account registry by `deploy all` itself (or `deploy images` ahead of it) over HTTPS and referenced from there — no Docker anywhere, a Cloudflare API token with Containers Edit in CLOUDFLARE_API_TOKEN for the copy (an installation deploying published images; what `init` writes from the published package); or `build` (the default when absent) — each Worker's Dockerfile, built by wrangler where `deploy all` runs (a checkout; the project's own production). `artifacts` (optional) names the R2 bucket a run's files move through (docs/reference/specs/execution.md item 20): with it the bot Worker binds the bucket and `deploy` creates it before the upload (the credential then needs Workers R2 Storage: Edit); the bot's runtime config `artifacts.r2.bucket` must say the same name. `metrics` (optional) names the Analytics Engine dataset every finished run's point is written to (docs/reference/specs/run-metrics.md): the state Worker's template then binds it as `RUN_METRICS` with the name beside it in `RUN_METRICS_DATASET` — nothing is created ahead of the deploy, the platform creates the dataset on first write — and the bot's runtime config `metrics.dataset` must say the same name (the bot warns at boot when the two differ). `deploy plan` reads this example when no profile exists; `deploy all` refuses it.",
3
3
  "account": "00000000000000000000000000000000",
4
4
  "zone": "example.com",
5
5
  "workers": {
@@ -9,5 +9,6 @@
9
9
  "sandbox": { "script": "switchboard-sandbox", "hostname": "switchboard-sandbox.example.com" }
10
10
  },
11
11
  "configSource": "config/config.yaml",
12
+ "restart": { "deployer": "deployer" },
12
13
  "images": "registry"
13
14
  }
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "switchboard",
3
- "version": "1.260.0",
3
+ "version": "1.260.2",
4
4
  "lockfileVersion": 3,
5
5
  "requires": true,
6
6
  "packages": {
7
7
  "": {
8
8
  "name": "switchboard",
9
- "version": "1.260.0",
9
+ "version": "1.260.2",
10
10
  "license": "Apache-2.0",
11
11
  "workspaces": [
12
12
  "web",
@@ -20445,7 +20445,7 @@
20445
20445
  },
20446
20446
  "packages/switchboard": {
20447
20447
  "name": "@coreplane/switchboard",
20448
- "version": "1.260.0",
20448
+ "version": "1.260.2",
20449
20449
  "license": "Apache-2.0",
20450
20450
  "dependencies": {
20451
20451
  "@earendil-works/pi-ai": "0.85.1",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "switchboard",
3
- "version": "1.260.0",
3
+ "version": "1.260.2",
4
4
  "private": true,
5
5
  "description": "Mention it in Slack and an agent reviews the PR, ships the fix, or answers the question — on the model you choose, with its tools running where you decide.",
6
6
  "license": "Apache-2.0",
@@ -1,5 +1,5 @@
1
1
  {
2
- "version": "1.260.0",
3
- "commit": "33c6cd103ce9d8a49be5249f98306fe46160f81c",
4
- "builtAt": "2026-09-21T15:19:23.063Z"
2
+ "version": "1.260.2",
3
+ "commit": "95c0153dc141f2e586238634e66970fea994bf7b",
4
+ "builtAt": "2026-09-21T20:25:49.397Z"
5
5
  }
@@ -447,7 +447,7 @@ Your final message is posted to Slack — keep it readable, lead with the outcom
447
447
  // whole reply — so the prose never repeats them and never pads around them.
448
448
  const REVIEW_FINAL_MESSAGE = `YOUR FINAL MESSAGE IS THE REVIEW'S TEXT, NOTHING ELSE. Switchboard renders the verdict line and the findings list from your submit_verdict call — on GitHub as the head of the comment, in Slack as the whole reply — and folds your final message under them on GitHub as the full review. So write only what the list cannot carry: one short paragraph per finding, keyed by its id (what is wrong, the concrete failure, the fix). Do not restate the verdict or the findings, do not summarize what you read, do not list what you verified clean, do not describe your method — what you checked belongs in your notes, which the run page shows. A change with no findings needs one sentence, not a tour.`;
449
449
 
450
- const REVIEW_VERDICT_INSTRUCTION = `VERDICT: before your final message, call the submit_verdict tool exactly once with \`approve\` (no finding at or above the severity to address remains — the level in force for this run, \`minor\` by default: a major or a minor finding means \`request_changes\`; nits alone never block) or \`request_changes\`, a one-line summary, \`head\` = the output of \`git rev-parse HEAD\` in the checkout you reviewed, and \`findings\` — every issue you report as a structured entry with a stable id you assign in order (F1, F2, …), a severity of exactly blocking|major|minor|nit, the file (plus line when it points at one), and a one-line title. The findings array is the index of your review: the full explanation of each finding stays in your prose, keyed by the same ids. Switchboard writes the verdict as the first line of the GitHub comment itself, lists the findings under it and folds your text below them as the full review; a review with no submitted verdict is posted as not approving, so never skip it. An \`approve\` carrying a finding at or above the severity to address is downgraded to \`request_changes\` and the tool's ack says so — approve only when every finding sits below the level. Do not write "LGTM" in your own text — the verdict line carries it.`;
450
+ const REVIEW_VERDICT_INSTRUCTION = `VERDICT: before your final message, call the submit_verdict tool exactly once with \`approve\` (no finding at or above the severity to address remains — the level in force for this run, \`minor\` by default: a major or a minor finding means \`request_changes\`; nits alone never block) or \`request_changes\`, a one-line summary, \`head\` = the output of \`git rev-parse HEAD\` in the checkout you reviewed, and \`findings\` — every issue you report as a structured entry with a stable id you assign in order (F1, F2, …), a severity of exactly blocking|major|minor|nit, the file (plus line when it points at one), and a one-line title. The findings array is the index of your review: the full explanation of each finding stays in your prose, keyed by the same ids. Switchboard writes the verdict as the first line of the GitHub comment itself, lists the findings under it and folds your text below them as the full review; a review with no submitted verdict is posted as not approving, so never skip it. The only exception is a preflight infrastructure failure from the REVIEW TARGET block's first-command HEAD check: stop, submit no finding or verdict, and let Switchboard report it without a GitHub post. An \`approve\` carrying a finding at or above the severity to address is downgraded to \`request_changes\` and the tool's ack says so — approve only when every finding sits below the level. Do not write "LGTM" in your own text — the verdict line carries it.`;
451
451
 
452
452
  // The diff-gated spec review (docs/reference/specs/agent-review.md item 14) and
453
453
  // the test guard under it (item 16; specs-coverage.md item 6), one text for
@@ -485,7 +485,7 @@ Strategy — GATHER ONCE, THEN ANALYZE ONCE. Do not explore file-by-file; your c
485
485
 
486
486
  1. GATHER, in 2-4 batched tool calls total:
487
487
  - \`gh pr view <ref> --json title,body,url,baseRefName\` and \`gh pr diff <ref>\` (the complete diff) in one command
488
- - clone the repo and check out the PR branch, then call the \`diff_digest\` tool to orient: per-file churn, totals, and risky-file flags (migrations/schema, auth/permission, whole-file deletions, lockfiles, very large files) so you know where to look hardest before you read a line
488
+ - for a PR review, use the checkout Switchboard provisioned at the head named in the REVIEW TARGET block — do not clone or check out another ref — then call the \`diff_digest\` tool to orient: per-file churn, totals, and risky-file flags (migrations/schema, auth/permission, whole-file deletions, lockfiles, very large files) so you know where to look hardest before you read a line
489
489
  ${REVIEW_WHOLE_CHANGE}
490
490
  - in ONE command, print the full current contents of every changed source file, e.g.: \`gh pr diff <ref> --name-only | grep -v -E "lock|generated|snap" | while read f; do echo "=== $f ==="; cat "$f"; done\`
491
491
  - if the PR is enormous (>~6k changed lines), print the riskiest files in full (state mutation, auth, concurrency, data deletion, public APIs) and only the diff hunks for the rest — and say which files you skimmed
@@ -104,11 +104,18 @@ export const HARNESS_PROBE_WAIT_MS = 5 * MINUTE_MS;
104
104
  * `maxMinutes` so one nobody lifted is an hour and a half, not a day. A run
105
105
  * asked during a drain waits at its attach one `pollMs` at a time under its
106
106
  * own lease less `leaseReserveMs` (what the attach and the work after it
107
- * need), `waitMaxMs` with no lease to clip it. */
107
+ * need), `waitMaxMs` with no lease to clip it — unless a fallback stands
108
+ * behind the attach: `fallbackWaitMs` bounds the wait by the FALLBACK's own
109
+ * cost, never the deploy's (issue 2101: a run refused by the drain waited
110
+ * 18 minutes for the whole deploy, then did the job in a seeded sandbox that
111
+ * stands up in about two), so a drain refusal on a first attach falls to the
112
+ * seeded sandbox after at most a few minutes; a re-attach with no fallback —
113
+ * a resumed run, a mid-run recovery — keeps the lease's bound. */
108
114
  export const DRAIN = {
109
115
  pollMs: 30_000,
110
116
  waitMaxMs: 60 * MINUTE_MS,
111
117
  leaseReserveMs: 10 * MINUTE_MS,
118
+ fallbackWaitMs: 3 * MINUTE_MS,
112
119
  deployWaitMaxMs: 60 * MINUTE_MS,
113
120
  marginMinutes: 5,
114
121
  maxMinutes: 90,
@@ -218,13 +225,19 @@ export const MERGE_WAIT_ASK_MINUTES = 60;
218
225
  * request's line with the conflict named, never a retry loop. */
219
226
  export const PULL_SWEEP = { leaseMinutes: 15, spendCapUsd: 5 } as const;
220
227
 
221
- /** The provider retry ladder (issue 1932): the backoff before each retry of a
222
- * transient model-call failure — a gateway 5xx, a stream cut before
223
- * `message_stop`, a gateway timeout. Three attempts with growing waits,
224
- * charged to the run's lease; the waits stay small next to the run's minutes
225
- * because the observed blips are edge transients of seconds. */
228
+ /** The provider retry ladder: the first backoffs for a transport-class
229
+ * model-call failure. The final rung repeats while the run's loop lease has
230
+ * time, so these are pacing intervals rather than a three-attempt terminal
231
+ * budget. */
226
232
  export const PROVIDER_RETRY_BACKOFFS_MS = [5_000, 15_000, 45_000] as const;
227
233
 
234
+ /** How often the model proxy writes an SSE comment while the provider is
235
+ * reasoning silently. The public container hop has cut an otherwise healthy
236
+ * stream at about thirty seconds of silence; a comment inside half that
237
+ * window keeps the transport alive without becoming a model event. The
238
+ * model call itself remains bounded by the run's lease. */
239
+ export const MODEL_STREAM_HEARTBEAT_MS = 15_000;
240
+
228
241
  /** A hosted ship parent's deadline margin past the pipeline's wall clock
229
242
  * (record 0060): the row's `state.hosting.until` is the hand-off time plus
230
243
  * the instance's `caps.maxMinutes` plus this hour, absorbing the runner's own