@coreplane/switchboard 1.260.0 → 1.260.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/switchboard.js +15 -11
- package/dist/assets/config/config.example.yaml +1 -0
- package/dist/assets/deploy/cloudflare/worker.ts +17 -56
- package/dist/assets/deploy/cloudflare/wrangler.template.jsonc +6 -0
- package/dist/assets/deploy/cloudflare-memory/backgroundTasks.ts +25 -0
- package/dist/assets/deploy/cloudflare-memory/worker.ts +36 -16
- package/dist/assets/deploy/cloudflare-resident/worker.ts +39 -18
- package/dist/assets/deploy/profile.example.json +2 -1
- package/dist/assets/package-lock.json +3 -3
- package/dist/assets/package.json +1 -1
- package/dist/assets/source.json +3 -3
- package/dist/assets/src/agents/registry.ts +2 -2
- package/dist/assets/src/core/budgets.ts +19 -6
- package/dist/assets/src/core/coordinator/driver.ts +47 -9
- package/dist/assets/src/core/pipelineStanding.ts +2 -0
- package/dist/assets/src/core/reviewedHead.ts +16 -2
- package/dist/assets/src/core/runEvents.ts +9 -1
- package/dist/assets/src/core/runLedger/types.ts +3 -3
- package/dist/assets/src/core/runRecord.ts +28 -8
- package/dist/assets/src/core/ship/coordinator.ts +248 -44
- package/dist/assets/src/core/ship/renewal.ts +2 -0
- package/dist/assets/src/deploy/liveGate.ts +8 -0
- package/dist/assets/src/deploy/profile.ts +8 -0
- package/dist/assets/src/deploy/restart.ts +49 -35
- package/dist/assets/web/dist/.vite/manifest.json +67 -67
- package/dist/assets/web/dist/assets/{DeliveryPage-DvMrWUg7.js → DeliveryPage-62qCSLvk.js} +1 -1
- package/dist/assets/web/dist/assets/{HomePage-CcGEJ4w0.js → HomePage-Dk3HRBc3.js} +1 -1
- package/dist/assets/web/dist/assets/{PendingTurnRow-BzGVxDYs.js → PendingTurnRow-CeTGOXUy.js} +1 -1
- package/dist/assets/web/dist/assets/{PlanePage-Arj9cyd5.js → PlanePage-RD0M6O6W.js} +1 -1
- package/dist/assets/web/dist/assets/{ResidentDetailPage-CBPPe4Sj.js → ResidentDetailPage-Dl2KId95.js} +1 -1
- package/dist/assets/web/dist/assets/{ResidentsIndexPage-CEpXgGo7.js → ResidentsIndexPage-DBy43I1R.js} +1 -1
- package/dist/assets/web/dist/assets/{RunFoldRow-BJq7tnLr.js → RunFoldRow-CwHDL-Mo.js} +1 -1
- package/dist/assets/web/dist/assets/{RunRoutePage-DWiny7LN.js → RunRoutePage-Du1HqLZs.js} +3 -3
- package/dist/assets/web/dist/assets/{RunsIndexPage-Cxufw5sV.js → RunsIndexPage-BZphy3Hr.js} +1 -1
- package/dist/assets/web/dist/assets/{ScheduledPage-SBcuDdVg.js → ScheduledPage-B7jh898t.js} +1 -1
- package/dist/assets/web/dist/assets/{SettingsPage-DrD_vWdp.js → SettingsPage-3uLEsiiE.js} +1 -1
- package/dist/assets/web/dist/assets/{SilentTurn-Bjl7EPEN.js → SilentTurn-CKxSxnQC.js} +1 -1
- package/dist/assets/web/dist/assets/{StatusDot-BqteQTav.js → StatusDot-x7vcK0iE.js} +1 -1
- package/dist/assets/web/dist/assets/{Tooltip-KRqJXEPN.js → Tooltip-wVXCWhMp.js} +1 -1
- package/dist/assets/web/dist/assets/{UnitRoutePage-Cv28FV4I.js → UnitRoutePage-V5BXK7FH.js} +1 -1
- package/dist/assets/web/dist/assets/budgets-KNNkT4PZ.js +1 -0
- package/dist/assets/web/dist/assets/{dist-CTASUSno.js → dist-C-XGnvM4.js} +1 -1
- package/dist/assets/web/dist/assets/{indexRow-B1YBy_IL.js → indexRow-Dxg7C8s9.js} +1 -1
- package/dist/assets/web/dist/assets/{main-eTAe9-hk.js → main-DAL--hyj.js} +2 -2
- package/dist/assets/web/dist/assets/sseReplay-BqgOggyQ.js +11 -0
- package/dist/cli.js +1405 -436
- package/package.json +1 -1
- package/dist/assets/web/dist/assets/budgets-DQXVllsm.js +0 -1
- package/dist/assets/web/dist/assets/sseReplay-BcsNbn2j.js +0 -11
package/bin/switchboard.js
CHANGED
|
@@ -1,25 +1,29 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
// The `switchboard` bin
|
|
3
|
-
// the
|
|
4
|
-
//
|
|
5
|
-
//
|
|
6
|
-
//
|
|
7
|
-
// the checkout — where npx prefers the workspace over the registry — found no
|
|
8
|
-
// link and died with `sh: switchboard: command not found`. The published package
|
|
9
|
-
// always carries the bundle, so there this is one extra module load; a checkout
|
|
10
|
-
// that has not built the package is told what to run.
|
|
2
|
+
// The published `switchboard` bin hands the process to the bundle beside it.
|
|
3
|
+
// In a repository checkout npm resolves the package name to this workspace too,
|
|
4
|
+
// but an ignored dist/ may be older than source. The checkout therefore never
|
|
5
|
+
// trusts dist: operators run the root source script, while a tarball runs only
|
|
6
|
+
// the bundle that its build packed.
|
|
11
7
|
import { existsSync } from "node:fs";
|
|
12
8
|
import { fileURLToPath } from "node:url";
|
|
13
9
|
|
|
14
10
|
const bundle = new URL("../dist/cli.js", import.meta.url);
|
|
11
|
+
const checkoutRoot = new URL("../../../", import.meta.url);
|
|
12
|
+
const runsFromCheckout =
|
|
13
|
+
existsSync(new URL("project.json", checkoutRoot)) && existsSync(new URL("src/cli.ts", checkoutRoot));
|
|
15
14
|
|
|
16
|
-
if (
|
|
15
|
+
if (runsFromCheckout) {
|
|
17
16
|
console.error(
|
|
18
|
-
|
|
17
|
+
"switchboard: npx resolved the package name to this checkout; its ignored dist/ is not authoritative. Run `npm run --silent cli -- <group> <verb> …` from the checkout root instead.",
|
|
19
18
|
);
|
|
20
19
|
process.exit(1);
|
|
21
20
|
}
|
|
22
21
|
|
|
22
|
+
if (!existsSync(bundle)) {
|
|
23
|
+
console.error(`switchboard: ${fileURLToPath(bundle)} is missing — the published package is incomplete.`);
|
|
24
|
+
process.exit(1);
|
|
25
|
+
}
|
|
26
|
+
|
|
23
27
|
// From here the bundle is the script: its entry claim (src/invokedAsScript.ts)
|
|
24
28
|
// and its usage spelling (`programName`, src/cli.ts) read argv[1].
|
|
25
29
|
process.argv[1] = fileURLToPath(bundle);
|
|
@@ -172,6 +172,7 @@ defaults:
|
|
|
172
172
|
# "config set channel ..." are stored in data/overrides.json and win over these.
|
|
173
173
|
# channels:
|
|
174
174
|
# slack:C012345:
|
|
175
|
+
# repo: acme/api # default repository for repo-bound asks in this channel
|
|
175
176
|
# agent: review
|
|
176
177
|
# models:
|
|
177
178
|
# review: anthropic/claude-opus-5
|
|
@@ -28,14 +28,10 @@ import {
|
|
|
28
28
|
import {
|
|
29
29
|
authenticateIngressBearer,
|
|
30
30
|
authenticateRestart,
|
|
31
|
+
authorizeRestartDeployer,
|
|
31
32
|
decideRestart,
|
|
32
|
-
parseRestartAuthorization,
|
|
33
33
|
parseRestartRequest,
|
|
34
|
-
RESTART_AUTHORIZE_PATH,
|
|
35
|
-
RESTART_SUBJECT_HEADER,
|
|
36
34
|
restartResponse,
|
|
37
|
-
stripRestartSubject,
|
|
38
|
-
type RestartAuth,
|
|
39
35
|
type RestartOutcome,
|
|
40
36
|
} from "../../src/deploy/restart.ts";
|
|
41
37
|
import {
|
|
@@ -89,7 +85,8 @@ export interface Env {
|
|
|
89
85
|
ACCESS_TEAM_DOMAIN?: string; // live-view SSO gate: Cloudflare Access team domain (JWKS + iss)
|
|
90
86
|
ACCESS_AUD?: string; // live-view SSO gate: Cloudflare Access application AUD tag
|
|
91
87
|
DASHBOARD_TOKEN?: string; // dashboard auth `token` strategy: the bearer (the default env name; config may name another)
|
|
92
|
-
SWITCHBOARD_INGRESS_TOKENS?: string; // enables HTTP /ingress + MCP /mcp (JSON token→identity map); the `cron` entry is what scheduled runs present
|
|
88
|
+
SWITCHBOARD_INGRESS_TOKENS?: string; // enables HTTP /ingress + MCP /mcp (JSON token→identity map); the `cron` entry is what scheduled runs present
|
|
89
|
+
SWITCHBOARD_RESTART_DEPLOYER?: string; // deployment-profile grant: the token-map subject allowed to POST /admin/restart; Worker-only, never runtime config
|
|
93
90
|
BRAVE_SEARCH_API_KEY?: string; // web_search backend (Brave); web_fetch works without it
|
|
94
91
|
GITHUB_WEBHOOK_SECRET?: string; // check-run intake: signs POST /webhooks/github; absent, the intake answers 503 disabled
|
|
95
92
|
CF_ANALYTICS_TOKEN?: string; // costs dash: Cloudflare API token, Account Analytics:Read only
|
|
@@ -211,37 +208,6 @@ export class SwitchboardServer extends Container<Env> {
|
|
|
211
208
|
return this.restartRunning(opts);
|
|
212
209
|
}
|
|
213
210
|
|
|
214
|
-
/**
|
|
215
|
-
* `deploy restart`, authorized: the Worker knows WHO the bearer is (the token
|
|
216
|
-
* map); WHETHER that identity may restart is the bot's config (`grants` —
|
|
217
|
-
* authorization.md item 9), which only the container holds. So ask it —
|
|
218
|
-
* `POST /admin/restart/authorize` with the authenticated subject in
|
|
219
|
-
* `RESTART_SUBJECT_HEADER` (never the bearer: the container may still hold the
|
|
220
|
-
* token map from before a rotation) — and stop only on a 200. A container that
|
|
221
|
-
* is not running is started first (the bot must answer): if the bearer is
|
|
222
|
-
* allowed, that start already put the current env live, so nothing is
|
|
223
|
-
* stopped and the outcome is `not-running`, exactly as before; if not, the
|
|
224
|
-
* refusal is relayed and the started container simply keeps serving.
|
|
225
|
-
*/
|
|
226
|
-
async restartAuthorized(
|
|
227
|
-
subject: string,
|
|
228
|
-
opts: { force: boolean },
|
|
229
|
-
): Promise<{ auth: RestartAuth; outcome?: RestartOutcome }> {
|
|
230
|
-
const wasRunning = this.ctx.container?.running === true;
|
|
231
|
-
await this.startBot();
|
|
232
|
-
const answer = await this.containerFetch(
|
|
233
|
-
new Request(`${INTERNAL}${RESTART_AUTHORIZE_PATH}`, {
|
|
234
|
-
method: "POST",
|
|
235
|
-
headers: { [RESTART_SUBJECT_HEADER]: subject },
|
|
236
|
-
}),
|
|
237
|
-
this.defaultPort,
|
|
238
|
-
);
|
|
239
|
-
const auth = parseRestartAuthorization(answer.status, await answer.text().catch(() => ""));
|
|
240
|
-
if (!auth.ok) return { auth };
|
|
241
|
-
if (!wasRunning) return { auth, outcome: { kind: "not-running" } };
|
|
242
|
-
return { auth, outcome: await this.restartRunning(opts) };
|
|
243
|
-
}
|
|
244
|
-
|
|
245
211
|
/** The stop itself, for a running container: the preflight's refusal rules over `/healthz`, then SIGTERM. */
|
|
246
212
|
private async restartRunning(opts: { force: boolean }): Promise<RestartOutcome> {
|
|
247
213
|
const health = await this.containerFetch(new Request(`${INTERNAL}/healthz`), this.defaultPort);
|
|
@@ -265,12 +231,12 @@ export class SwitchboardServer extends Container<Env> {
|
|
|
265
231
|
}
|
|
266
232
|
}
|
|
267
233
|
|
|
268
|
-
/** `POST /admin/restart` — the operator surface behind `deploy restart
|
|
269
|
-
*
|
|
270
|
-
*
|
|
271
|
-
*
|
|
272
|
-
*
|
|
273
|
-
*
|
|
234
|
+
/** `POST /admin/restart` — the operator surface behind `deploy restart`.
|
|
235
|
+
* The Worker authenticates the bearer against its token map, then authorizes
|
|
236
|
+
* that subject against SWITCHBOARD_RESTART_DEPLOYER, rendered from the
|
|
237
|
+
* deployment profile (or set directly as a Worker var). Runtime config is
|
|
238
|
+
* never consulted: this route must remain usable when that document is what
|
|
239
|
+
* the restart is repairing.
|
|
274
240
|
* Body `{ "force": true }` bypasses the fail-closed refusals (no JSON body,
|
|
275
241
|
* impossible `inFlight`); runs in flight or a drain warn and never refuse. */
|
|
276
242
|
async function handleAdminRestart(request: Request, env: Env): Promise<Response> {
|
|
@@ -283,14 +249,16 @@ async function handleAdminRestart(request: Request, env: Env): Promise<Response>
|
|
|
283
249
|
console.warn(`[restart] ${authn.status} — ${authn.reason}`);
|
|
284
250
|
return json(authn.status, { ok: false, error: authn.reason });
|
|
285
251
|
}
|
|
252
|
+
const auth = authorizeRestartDeployer(authn.identity.subject, env.SWITCHBOARD_RESTART_DEPLOYER);
|
|
253
|
+
if (!auth.ok) {
|
|
254
|
+
console.warn(`[restart] ${auth.status} — ${auth.reason}`);
|
|
255
|
+
return json(auth.status, { ok: false, error: auth.reason });
|
|
256
|
+
}
|
|
286
257
|
const parsed = parseRestartRequest(await request.text().catch(() => ""));
|
|
287
258
|
if (!parsed.ok) return json(400, { ok: false, error: parsed.reason });
|
|
288
|
-
let
|
|
289
|
-
let outcome: RestartOutcome | undefined;
|
|
259
|
+
let outcome: RestartOutcome;
|
|
290
260
|
try {
|
|
291
|
-
|
|
292
|
-
force: parsed.force,
|
|
293
|
-
}));
|
|
261
|
+
outcome = await getContainer(env.SWITCHBOARD, INSTANCE).restart({ force: parsed.force });
|
|
294
262
|
} catch (err) {
|
|
295
263
|
// The container's /healthz probe or the DO call threw (container mid-transition,
|
|
296
264
|
// port not answering): fail closed in the route's own JSON shape so the CLI reads
|
|
@@ -299,11 +267,6 @@ async function handleAdminRestart(request: Request, env: Env): Promise<Response>
|
|
|
299
267
|
console.error(`[restart] ${authn.identity.subject} → error before stop: ${reason}`);
|
|
300
268
|
return json(500, { ok: false, error: `restart failed before stopping anything: ${reason}` });
|
|
301
269
|
}
|
|
302
|
-
if (!auth.ok || outcome === undefined) {
|
|
303
|
-
const refusal = auth.ok ? { status: 503 as const, reason: "restart disabled: no outcome" } : auth;
|
|
304
|
-
console.warn(`[restart] ${refusal.status} — ${refusal.reason}`);
|
|
305
|
-
return json(refusal.status, { ok: false, error: refusal.reason });
|
|
306
|
-
}
|
|
307
270
|
console.log(
|
|
308
271
|
`[restart] ${auth.subject} → ${outcome.kind}${outcome.kind === "refused" ? `: ${outcome.problems.join("; ")}` : ""}`,
|
|
309
272
|
);
|
|
@@ -494,9 +457,7 @@ export default {
|
|
|
494
457
|
// caller sent is stripped, and what the container sees carries this
|
|
495
458
|
// Worker's own root. A static asset or the live view's SSE stream gets no
|
|
496
459
|
// root; a refusal's line is dropped by the sink's filter.
|
|
497
|
-
|
|
498
|
-
// authorize call below); a caller cannot be allowed to speak it.
|
|
499
|
-
const inbound = stripRestartSubject(stripTraceContext(request));
|
|
460
|
+
const inbound = stripTraceContext(request);
|
|
500
461
|
const route = shimRoute(pathname);
|
|
501
462
|
if (route === undefined) return withLength(await getContainer(env.SWITCHBOARD, INSTANCE).fetch(inbound));
|
|
502
463
|
const root = tracer.start("bot-shim.fetch", { sinks: traceSinks, attrs: { route } });
|
|
@@ -31,6 +31,12 @@
|
|
|
31
31
|
"ACCESS_TEAM_DOMAIN": "{{access.teamDomain}}",
|
|
32
32
|
"ACCESS_AUD": "{{access.aud}}",
|
|
33
33
|
// {{/if}}
|
|
34
|
+
// The restart route's deployment-side grant. The Worker authenticates a
|
|
35
|
+
// bearer through SWITCHBOARD_INGRESS_TOKENS and compares its subject here;
|
|
36
|
+
// it never asks the runtime config it may be restarting to repair.
|
|
37
|
+
// {{#if restart}}
|
|
38
|
+
"SWITCHBOARD_RESTART_DEPLOYER": "{{restart.deployer}}",
|
|
39
|
+
// {{/if}}
|
|
34
40
|
// The state Worker (deploy/cloudflare-memory/): where worker.ts records each
|
|
35
41
|
// scheduled firing for the /runs Scheduled panel, with MEMORY_TOKEN.
|
|
36
42
|
// A profile without a memory Worker renders no var: firings go unrecorded.
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
type WaitUntilContext = Pick<DurableObjectState, "waitUntil">;
|
|
2
|
+
|
|
3
|
+
const TASKS = Symbol.for("switchboard.memory-worker.pending-background-tasks");
|
|
4
|
+
type PendingTasks = Map<Promise<unknown>, string>;
|
|
5
|
+
|
|
6
|
+
const pendingTasks = (): PendingTasks => {
|
|
7
|
+
const root = globalThis as typeof globalThis & { [TASKS]?: PendingTasks };
|
|
8
|
+
return (root[TASKS] ??= new Map());
|
|
9
|
+
};
|
|
10
|
+
|
|
11
|
+
/** Keep deliberately detached Worker work in the runtime's request lifetime.
|
|
12
|
+
* The process-wide label set is also the test seam: a case cannot finish while
|
|
13
|
+
* work it started is still able to contend with the next case. */
|
|
14
|
+
export function holdBackgroundTask(context: WaitUntilContext, label: string, task: Promise<unknown>): void {
|
|
15
|
+
const pending = pendingTasks();
|
|
16
|
+
const held = task.finally(() => pending.delete(held));
|
|
17
|
+
pending.set(held, label);
|
|
18
|
+
context.waitUntil(held);
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
export function assertNoPendingBackgroundTasks(): void {
|
|
22
|
+
const labels = [...pendingTasks().values()];
|
|
23
|
+
if (labels.length === 0) return;
|
|
24
|
+
throw new Error(`test ended with ${labels.length} pending background task(s): ${labels.join(", ")}`);
|
|
25
|
+
}
|
|
@@ -78,6 +78,7 @@ import {
|
|
|
78
78
|
selectReclaim,
|
|
79
79
|
} from "../../src/core/runLedger/decisions.ts";
|
|
80
80
|
import { intakeReceiptRetentionMs, minutesToMs, PLANE } from "../../src/core/budgets.ts";
|
|
81
|
+
import { holdBackgroundTask } from "./backgroundTasks.ts";
|
|
81
82
|
import {
|
|
82
83
|
causeOfClose,
|
|
83
84
|
causeOfReclaim,
|
|
@@ -2058,7 +2059,7 @@ export class RunHistoryDO extends DurableObject<Env> {
|
|
|
2058
2059
|
/** The admission ask (`POST /plane/admit`, record 0064 "The queue"): one
|
|
2059
2060
|
* transaction decides and writes — `admitted` reserves the thread,
|
|
2060
2061
|
* `queued` stores the request under the minted id. */
|
|
2061
|
-
planeAdmit(
|
|
2062
|
+
async planeAdmit(
|
|
2062
2063
|
post: {
|
|
2063
2064
|
runId: string;
|
|
2064
2065
|
requester: string;
|
|
@@ -2070,7 +2071,7 @@ export class RunHistoryDO extends DurableObject<Env> {
|
|
|
2070
2071
|
reaskMs?: number;
|
|
2071
2072
|
},
|
|
2072
2073
|
now: number,
|
|
2073
|
-
): PlaneAskAnswer {
|
|
2074
|
+
): Promise<PlaneAskAnswer> {
|
|
2074
2075
|
let answer!: PlaneAskAnswer;
|
|
2075
2076
|
this.ctx.storage.transactionSync(() => {
|
|
2076
2077
|
if (post.reaskMs !== undefined)
|
|
@@ -2089,7 +2090,7 @@ export class RunHistoryDO extends DurableObject<Env> {
|
|
|
2089
2090
|
this.applyPlaneWrites(decision.writes);
|
|
2090
2091
|
answer = planeAskAnswerOf(decision, post.runId);
|
|
2091
2092
|
});
|
|
2092
|
-
if (answer.kind === "queued")
|
|
2093
|
+
if (answer.kind === "queued") await this.settlePlaneReaskAlarmAfterCommit(now);
|
|
2093
2094
|
console.log(
|
|
2094
2095
|
`[plane/admit] ${post.threadKey} → ${answer.kind}${answer.kind === "queued" ? ` position ${answer.position}` : ""} (run ${post.runId})`,
|
|
2095
2096
|
);
|
|
@@ -2152,7 +2153,10 @@ export class RunHistoryDO extends DurableObject<Env> {
|
|
|
2152
2153
|
|
|
2153
2154
|
/** A refusal-by-name the bot met at attach or exec (`POST /plane/observe`,
|
|
2154
2155
|
* record 0064): an admitted run re-enters the queue at its old position. */
|
|
2155
|
-
planeObserve(
|
|
2156
|
+
async planeObserve(
|
|
2157
|
+
post: { runId: string; resident: string; refusal: string },
|
|
2158
|
+
now: number,
|
|
2159
|
+
): Promise<{ reentered: boolean }> {
|
|
2156
2160
|
let reentered = false;
|
|
2157
2161
|
this.ctx.storage.transactionSync(() => {
|
|
2158
2162
|
const decision = decide(this.planeState(), {
|
|
@@ -2165,7 +2169,7 @@ export class RunHistoryDO extends DurableObject<Env> {
|
|
|
2165
2169
|
this.applyPlaneWrites(decision.writes);
|
|
2166
2170
|
reentered = decision.writes.length > 0;
|
|
2167
2171
|
});
|
|
2168
|
-
if (reentered)
|
|
2172
|
+
if (reentered) await this.settlePlaneReaskAlarmAfterCommit(now);
|
|
2169
2173
|
console.log(
|
|
2170
2174
|
`[plane/observe] run ${post.runId} on ${post.resident}: ${post.refusal.slice(0, 60)} — ${reentered ? "re-entered" : "no-op"}`,
|
|
2171
2175
|
);
|
|
@@ -2194,6 +2198,18 @@ export class RunHistoryDO extends DurableObject<Env> {
|
|
|
2194
2198
|
if (set === null || set > due) await this.ctx.storage.setAlarm(due);
|
|
2195
2199
|
}
|
|
2196
2200
|
|
|
2201
|
+
/** Settle alarm I/O without contradicting the plane transaction that already
|
|
2202
|
+
* committed. A scheduling failure may delay a re-ask until another wake,
|
|
2203
|
+
* but returning an error would make the caller proceed or retry while the
|
|
2204
|
+
* durable queued row remains eligible for admission. */
|
|
2205
|
+
private async settlePlaneReaskAlarmAfterCommit(now: number): Promise<void> {
|
|
2206
|
+
try {
|
|
2207
|
+
await this.ensurePlaneReaskAlarm(now);
|
|
2208
|
+
} catch (err) {
|
|
2209
|
+
console.error("[plane/alarm] re-ask scheduling failed after the plane state committed", err);
|
|
2210
|
+
}
|
|
2211
|
+
}
|
|
2212
|
+
|
|
2197
2213
|
/** The earliest instant the plane must wake at (record 0064): the
|
|
2198
2214
|
* earliest hosting deadline, lease end or re-ask across its rows. A due
|
|
2199
2215
|
* already past re-offers at the re-ask cadence, never in a hot loop, and
|
|
@@ -2372,21 +2388,24 @@ export class RunHistoryDO extends DurableObject<Env> {
|
|
|
2372
2388
|
|
|
2373
2389
|
/** The transport's push (record 0064, "Where it lives"): committed effects
|
|
2374
2390
|
* are pushed to the bot Worker over the service binding, which forwards to
|
|
2375
|
-
* the container.
|
|
2376
|
-
*
|
|
2391
|
+
* the container. The response does not wait for this best-effort push, but
|
|
2392
|
+
* the actor does: waitUntil keeps its I/O inside this request's lifetime so
|
|
2393
|
+
* it cannot contend with an unrelated request after the caller moves on. */
|
|
2377
2394
|
private pushPlaneEffects(effects: PlaneEffect[]): void {
|
|
2378
2395
|
if (effects.length === 0) return;
|
|
2379
2396
|
const bot = this.env.BOT;
|
|
2380
2397
|
if (!bot) return;
|
|
2381
|
-
|
|
2382
|
-
.
|
|
2383
|
-
|
|
2384
|
-
|
|
2385
|
-
|
|
2386
|
-
|
|
2387
|
-
|
|
2388
|
-
|
|
2389
|
-
|
|
2398
|
+
const delivery = Promise.resolve()
|
|
2399
|
+
.then(() =>
|
|
2400
|
+
bot.fetch("https://bot/plane/effects", {
|
|
2401
|
+
method: "POST",
|
|
2402
|
+
headers: {
|
|
2403
|
+
"content-type": "application/json",
|
|
2404
|
+
authorization: `Bearer ${this.env.MEMORY_TOKEN ?? ""}`,
|
|
2405
|
+
},
|
|
2406
|
+
body: JSON.stringify({ effects }),
|
|
2407
|
+
}),
|
|
2408
|
+
)
|
|
2390
2409
|
.then((r) => {
|
|
2391
2410
|
if (!r.ok)
|
|
2392
2411
|
console.warn(
|
|
@@ -2398,6 +2417,7 @@ export class RunHistoryDO extends DurableObject<Env> {
|
|
|
2398
2417
|
`[plane/push] ${effects.length} effect(s) not delivered: ${err instanceof Error ? err.message : String(err)} — they ride the next heartbeat`,
|
|
2399
2418
|
);
|
|
2400
2419
|
});
|
|
2420
|
+
holdBackgroundTask(this.ctx, `plane effect push (${effects.map((effect) => effect.id).join(", ")})`, delivery);
|
|
2401
2421
|
}
|
|
2402
2422
|
|
|
2403
2423
|
/** The unacknowledged effects, oldest first, at most `PLANE_EFFECTS_PER_ANSWER`
|
|
@@ -5027,11 +5027,32 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5027
5027
|
* deploy's alone. Answers `restarted` when a stop was issued, else why
|
|
5028
5028
|
* not. */
|
|
5029
5029
|
private async reconcileImage(where: "attach" | "refresh" | "deploy", force = false): Promise<ImageReconcileResult> {
|
|
5030
|
-
|
|
5031
|
-
|
|
5032
|
-
|
|
5033
|
-
|
|
5034
|
-
|
|
5030
|
+
const active = await this.isRuntimeActive().catch((err) => {
|
|
5031
|
+
console.log(`image-reconcile (${where}): activity probe failed — preserving the pending report: ${errMsg(err)}`);
|
|
5032
|
+
return null;
|
|
5033
|
+
});
|
|
5034
|
+
// A failed probe proves neither state. Preserve the marker and its drain
|
|
5035
|
+
// hold until a later reconcile confirms inactivity or hydrates a fresh
|
|
5036
|
+
// active container. An attach with that marker must fail closed even after
|
|
5037
|
+
// the drain backstop reopens: admitting it could wake a pre-deploy image.
|
|
5038
|
+
if (active === null) {
|
|
5039
|
+
if (
|
|
5040
|
+
where === "attach" &&
|
|
5041
|
+
(await this.ctx.storage.get<{ resource: string }>(IMAGE_REPORT_PENDING_KEY)) !== undefined
|
|
5042
|
+
) {
|
|
5043
|
+
console.log(
|
|
5044
|
+
"image-stale (attach): activity could not be verified while a new-image report is pending — the attach is refused",
|
|
5045
|
+
);
|
|
5046
|
+
return "stale";
|
|
5047
|
+
}
|
|
5048
|
+
return "deferred";
|
|
5049
|
+
}
|
|
5050
|
+
if (!active) {
|
|
5051
|
+
// An inactive container's next start is on the deployed image by
|
|
5052
|
+
// construction. Report on every reconcile: the deploy may have marked
|
|
5053
|
+
// it pending after an earlier idle cycle, and the drain itself refuses
|
|
5054
|
+
// the attach that would otherwise wake it and provide another reporter.
|
|
5055
|
+
await this.reportPendingImageCurrent(where);
|
|
5035
5056
|
return "inactive";
|
|
5036
5057
|
}
|
|
5037
5058
|
const last = THREAD_USERS[THREAD_USERS.length - 1];
|
|
@@ -5091,12 +5112,13 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5091
5112
|
/** The report a held drain waits for (issue 1931): when the deploy's
|
|
5092
5113
|
* reconcile could not verify this resident's fresh container on the new
|
|
5093
5114
|
* image, a marker stays in storage and the registry holds the drain. The
|
|
5094
|
-
* fact the report stands on is
|
|
5095
|
-
* the
|
|
5096
|
-
*
|
|
5097
|
-
*
|
|
5098
|
-
* a transient failure keeps it for
|
|
5099
|
-
* `until` is the backstop for a report
|
|
5115
|
+
* fact the report stands on is either an inactive runtime, whose next start
|
|
5116
|
+
* uses the deployed image by construction, or the active replacement's own
|
|
5117
|
+
* completed hydration: the fresh container has restored its snapshot and
|
|
5118
|
+
* reached `warm`. A stop alone does not report (issue 2101). The marker is
|
|
5119
|
+
* deleted only after a successful report: a transient failure keeps it for
|
|
5120
|
+
* the next reconcile, and the drain's `until` is the backstop for a report
|
|
5121
|
+
* that never lands. */
|
|
5100
5122
|
private async reportPendingImageCurrent(where: string): Promise<boolean> {
|
|
5101
5123
|
const pending = await this.ctx.storage.get<{ resource: string }>(IMAGE_REPORT_PENDING_KEY);
|
|
5102
5124
|
if (pending === undefined) return true;
|
|
@@ -5125,9 +5147,9 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5125
5147
|
* an active, quiet container is always cycled, but that stop is not a report.
|
|
5126
5148
|
* The marker survives until the replacement hydrates and reaches `warm` on
|
|
5127
5149
|
* the deploy's image (issue 2101: the old probe raced that restore and lost).
|
|
5128
|
-
* `verified` true is the fact the reopen may stand on
|
|
5129
|
-
*
|
|
5130
|
-
* `reportPendingImageCurrent`, never
|
|
5150
|
+
* `verified` true is the fact the reopen may stand on: inactivity reports
|
|
5151
|
+
* at once, while an active container's cycle leaves a hold until its fresh
|
|
5152
|
+
* hydration calls `reportPendingImageCurrent`, never until a timer.
|
|
5131
5153
|
*
|
|
5132
5154
|
* The hold lands BEFORE the marker: the marker is what lets any concurrent
|
|
5133
5155
|
* attach/refresh reconcile report, and a report that reaches the registry
|
|
@@ -5139,10 +5161,9 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5139
5161
|
await this.ctx.storage.put(IMAGE_REPORT_PENDING_KEY, { resource });
|
|
5140
5162
|
const result = await this.reconcileImage("deploy", true);
|
|
5141
5163
|
if (result === "deferred") return { result, verified: false };
|
|
5142
|
-
//
|
|
5143
|
-
//
|
|
5144
|
-
//
|
|
5145
|
-
// read rather than an unconditional false.
|
|
5164
|
+
// Inactivity reports synchronously; a successful stop does not. A
|
|
5165
|
+
// concurrent fresh start may also have hydrated while this reconcile
|
|
5166
|
+
// yielded, hence the storage read rather than a verdict from `result`.
|
|
5146
5167
|
const verified = (await this.ctx.storage.get(IMAGE_REPORT_PENDING_KEY)) === undefined;
|
|
5147
5168
|
return { result, verified };
|
|
5148
5169
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"$comment": "The deployment profile: where THIS installation runs. Copy to deploy/profile.json and fill it in (or run `switchboard deploy init`). The account is your Cloudflare account id; every hostname must be under `zone`, a zone in that account, unless the Worker names its own `zone` (also in the account); `workers.bot` is the one required Worker — leave `memory`, `resident` or `sandbox` out and `deploy plan` has no step for them (a bot-only profile is a one-step plan; without `memory` the config is not pushed anywhere). The project's docs site is not a Worker of an installation: it is the project's website, deployed by the project's own CI from project.json, so there is no `docs` entry. `configSource` is where `deploy all` reads the bot's runtime config from before building the image — a path, `github://owner/repo/path@ref` (needs CONFIG_REPO_TOKEN), or `op://Vault/Item/field` (needs OP_SERVICE_ACCOUNT_TOKEN); `secretsSource` is where `secrets put` reads values from — a directory of <NAME> files (the default when absent) or `op://Vault/Item`. `images` is where the bot, resident and sandbox container images come from: `registry` — the release's published images, copied once per version into your account registry by `deploy all` itself (or `deploy images` ahead of it) over HTTPS and referenced from there — no Docker anywhere, a Cloudflare API token with Containers Edit in CLOUDFLARE_API_TOKEN for the copy (an installation deploying published images; what `init` writes from the published package); or `build` (the default when absent) — each Worker's Dockerfile, built by wrangler where `deploy all` runs (a checkout; the project's own production). `artifacts` (optional) names the R2 bucket a run's files move through (docs/reference/specs/execution.md item 20): with it the bot Worker binds the bucket and `deploy` creates it before the upload (the credential then needs Workers R2 Storage: Edit); the bot's runtime config `artifacts.r2.bucket` must say the same name. `metrics` (optional) names the Analytics Engine dataset every finished run's point is written to (docs/reference/specs/run-metrics.md): the state Worker's template then binds it as `RUN_METRICS` with the name beside it in `RUN_METRICS_DATASET` — nothing is created ahead of the deploy, the platform creates the dataset on first write — and the bot's runtime config `metrics.dataset` must say the same name (the bot warns at boot when the two differ). `deploy plan` reads this example when no profile exists; `deploy all` refuses it.",
|
|
2
|
+
"$comment": "The deployment profile: where THIS installation runs. Copy to deploy/profile.json and fill it in (or run `switchboard deploy init`). The account is your Cloudflare account id; every hostname must be under `zone`, a zone in that account, unless the Worker names its own `zone` (also in the account); `workers.bot` is the one required Worker — leave `memory`, `resident` or `sandbox` out and `deploy plan` has no step for them (a bot-only profile is a one-step plan; without `memory` the config is not pushed anywhere). The project's docs site is not a Worker of an installation: it is the project's website, deployed by the project's own CI from project.json, so there is no `docs` entry. `configSource` is where `deploy all` reads the bot's runtime config from before building the image — a path, `github://owner/repo/path@ref` (needs CONFIG_REPO_TOKEN), or `op://Vault/Item/field` (needs OP_SERVICE_ACCOUNT_TOKEN); `restart.deployer` is the SWITCHBOARD_INGRESS_TOKENS subject the Worker itself allows to restart the bot, independent of runtime config so config recovery cannot lock itself out; `secretsSource` is where `secrets put` reads values from — a directory of <NAME> files (the default when absent) or `op://Vault/Item`. `images` is where the bot, resident and sandbox container images come from: `registry` — the release's published images, copied once per version into your account registry by `deploy all` itself (or `deploy images` ahead of it) over HTTPS and referenced from there — no Docker anywhere, a Cloudflare API token with Containers Edit in CLOUDFLARE_API_TOKEN for the copy (an installation deploying published images; what `init` writes from the published package); or `build` (the default when absent) — each Worker's Dockerfile, built by wrangler where `deploy all` runs (a checkout; the project's own production). `artifacts` (optional) names the R2 bucket a run's files move through (docs/reference/specs/execution.md item 20): with it the bot Worker binds the bucket and `deploy` creates it before the upload (the credential then needs Workers R2 Storage: Edit); the bot's runtime config `artifacts.r2.bucket` must say the same name. `metrics` (optional) names the Analytics Engine dataset every finished run's point is written to (docs/reference/specs/run-metrics.md): the state Worker's template then binds it as `RUN_METRICS` with the name beside it in `RUN_METRICS_DATASET` — nothing is created ahead of the deploy, the platform creates the dataset on first write — and the bot's runtime config `metrics.dataset` must say the same name (the bot warns at boot when the two differ). `deploy plan` reads this example when no profile exists; `deploy all` refuses it.",
|
|
3
3
|
"account": "00000000000000000000000000000000",
|
|
4
4
|
"zone": "example.com",
|
|
5
5
|
"workers": {
|
|
@@ -9,5 +9,6 @@
|
|
|
9
9
|
"sandbox": { "script": "switchboard-sandbox", "hostname": "switchboard-sandbox.example.com" }
|
|
10
10
|
},
|
|
11
11
|
"configSource": "config/config.yaml",
|
|
12
|
+
"restart": { "deployer": "deployer" },
|
|
12
13
|
"images": "registry"
|
|
13
14
|
}
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "switchboard",
|
|
3
|
-
"version": "1.260.
|
|
3
|
+
"version": "1.260.2",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "switchboard",
|
|
9
|
-
"version": "1.260.
|
|
9
|
+
"version": "1.260.2",
|
|
10
10
|
"license": "Apache-2.0",
|
|
11
11
|
"workspaces": [
|
|
12
12
|
"web",
|
|
@@ -20445,7 +20445,7 @@
|
|
|
20445
20445
|
},
|
|
20446
20446
|
"packages/switchboard": {
|
|
20447
20447
|
"name": "@coreplane/switchboard",
|
|
20448
|
-
"version": "1.260.
|
|
20448
|
+
"version": "1.260.2",
|
|
20449
20449
|
"license": "Apache-2.0",
|
|
20450
20450
|
"dependencies": {
|
|
20451
20451
|
"@earendil-works/pi-ai": "0.85.1",
|
package/dist/assets/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "switchboard",
|
|
3
|
-
"version": "1.260.
|
|
3
|
+
"version": "1.260.2",
|
|
4
4
|
"private": true,
|
|
5
5
|
"description": "Mention it in Slack and an agent reviews the PR, ships the fix, or answers the question — on the model you choose, with its tools running where you decide.",
|
|
6
6
|
"license": "Apache-2.0",
|
package/dist/assets/source.json
CHANGED
|
@@ -447,7 +447,7 @@ Your final message is posted to Slack — keep it readable, lead with the outcom
|
|
|
447
447
|
// whole reply — so the prose never repeats them and never pads around them.
|
|
448
448
|
const REVIEW_FINAL_MESSAGE = `YOUR FINAL MESSAGE IS THE REVIEW'S TEXT, NOTHING ELSE. Switchboard renders the verdict line and the findings list from your submit_verdict call — on GitHub as the head of the comment, in Slack as the whole reply — and folds your final message under them on GitHub as the full review. So write only what the list cannot carry: one short paragraph per finding, keyed by its id (what is wrong, the concrete failure, the fix). Do not restate the verdict or the findings, do not summarize what you read, do not list what you verified clean, do not describe your method — what you checked belongs in your notes, which the run page shows. A change with no findings needs one sentence, not a tour.`;
|
|
449
449
|
|
|
450
|
-
const REVIEW_VERDICT_INSTRUCTION = `VERDICT: before your final message, call the submit_verdict tool exactly once with \`approve\` (no finding at or above the severity to address remains — the level in force for this run, \`minor\` by default: a major or a minor finding means \`request_changes\`; nits alone never block) or \`request_changes\`, a one-line summary, \`head\` = the output of \`git rev-parse HEAD\` in the checkout you reviewed, and \`findings\` — every issue you report as a structured entry with a stable id you assign in order (F1, F2, …), a severity of exactly blocking|major|minor|nit, the file (plus line when it points at one), and a one-line title. The findings array is the index of your review: the full explanation of each finding stays in your prose, keyed by the same ids. Switchboard writes the verdict as the first line of the GitHub comment itself, lists the findings under it and folds your text below them as the full review; a review with no submitted verdict is posted as not approving, so never skip it. An \`approve\` carrying a finding at or above the severity to address is downgraded to \`request_changes\` and the tool's ack says so — approve only when every finding sits below the level. Do not write "LGTM" in your own text — the verdict line carries it.`;
|
|
450
|
+
const REVIEW_VERDICT_INSTRUCTION = `VERDICT: before your final message, call the submit_verdict tool exactly once with \`approve\` (no finding at or above the severity to address remains — the level in force for this run, \`minor\` by default: a major or a minor finding means \`request_changes\`; nits alone never block) or \`request_changes\`, a one-line summary, \`head\` = the output of \`git rev-parse HEAD\` in the checkout you reviewed, and \`findings\` — every issue you report as a structured entry with a stable id you assign in order (F1, F2, …), a severity of exactly blocking|major|minor|nit, the file (plus line when it points at one), and a one-line title. The findings array is the index of your review: the full explanation of each finding stays in your prose, keyed by the same ids. Switchboard writes the verdict as the first line of the GitHub comment itself, lists the findings under it and folds your text below them as the full review; a review with no submitted verdict is posted as not approving, so never skip it. The only exception is a preflight infrastructure failure from the REVIEW TARGET block's first-command HEAD check: stop, submit no finding or verdict, and let Switchboard report it without a GitHub post. An \`approve\` carrying a finding at or above the severity to address is downgraded to \`request_changes\` and the tool's ack says so — approve only when every finding sits below the level. Do not write "LGTM" in your own text — the verdict line carries it.`;
|
|
451
451
|
|
|
452
452
|
// The diff-gated spec review (docs/reference/specs/agent-review.md item 14) and
|
|
453
453
|
// the test guard under it (item 16; specs-coverage.md item 6), one text for
|
|
@@ -485,7 +485,7 @@ Strategy — GATHER ONCE, THEN ANALYZE ONCE. Do not explore file-by-file; your c
|
|
|
485
485
|
|
|
486
486
|
1. GATHER, in 2-4 batched tool calls total:
|
|
487
487
|
- \`gh pr view <ref> --json title,body,url,baseRefName\` and \`gh pr diff <ref>\` (the complete diff) in one command
|
|
488
|
-
-
|
|
488
|
+
- for a PR review, use the checkout Switchboard provisioned at the head named in the REVIEW TARGET block — do not clone or check out another ref — then call the \`diff_digest\` tool to orient: per-file churn, totals, and risky-file flags (migrations/schema, auth/permission, whole-file deletions, lockfiles, very large files) so you know where to look hardest before you read a line
|
|
489
489
|
${REVIEW_WHOLE_CHANGE}
|
|
490
490
|
- in ONE command, print the full current contents of every changed source file, e.g.: \`gh pr diff <ref> --name-only | grep -v -E "lock|generated|snap" | while read f; do echo "=== $f ==="; cat "$f"; done\`
|
|
491
491
|
- if the PR is enormous (>~6k changed lines), print the riskiest files in full (state mutation, auth, concurrency, data deletion, public APIs) and only the diff hunks for the rest — and say which files you skimmed
|
|
@@ -104,11 +104,18 @@ export const HARNESS_PROBE_WAIT_MS = 5 * MINUTE_MS;
|
|
|
104
104
|
* `maxMinutes` so one nobody lifted is an hour and a half, not a day. A run
|
|
105
105
|
* asked during a drain waits at its attach one `pollMs` at a time under its
|
|
106
106
|
* own lease less `leaseReserveMs` (what the attach and the work after it
|
|
107
|
-
* need), `waitMaxMs` with no lease to clip it
|
|
107
|
+
* need), `waitMaxMs` with no lease to clip it — unless a fallback stands
|
|
108
|
+
* behind the attach: `fallbackWaitMs` bounds the wait by the FALLBACK's own
|
|
109
|
+
* cost, never the deploy's (issue 2101: a run refused by the drain waited
|
|
110
|
+
* 18 minutes for the whole deploy, then did the job in a seeded sandbox that
|
|
111
|
+
* stands up in about two), so a drain refusal on a first attach falls to the
|
|
112
|
+
* seeded sandbox after at most a few minutes; a re-attach with no fallback —
|
|
113
|
+
* a resumed run, a mid-run recovery — keeps the lease's bound. */
|
|
108
114
|
export const DRAIN = {
|
|
109
115
|
pollMs: 30_000,
|
|
110
116
|
waitMaxMs: 60 * MINUTE_MS,
|
|
111
117
|
leaseReserveMs: 10 * MINUTE_MS,
|
|
118
|
+
fallbackWaitMs: 3 * MINUTE_MS,
|
|
112
119
|
deployWaitMaxMs: 60 * MINUTE_MS,
|
|
113
120
|
marginMinutes: 5,
|
|
114
121
|
maxMinutes: 90,
|
|
@@ -218,13 +225,19 @@ export const MERGE_WAIT_ASK_MINUTES = 60;
|
|
|
218
225
|
* request's line with the conflict named, never a retry loop. */
|
|
219
226
|
export const PULL_SWEEP = { leaseMinutes: 15, spendCapUsd: 5 } as const;
|
|
220
227
|
|
|
221
|
-
/** The provider retry ladder
|
|
222
|
-
*
|
|
223
|
-
*
|
|
224
|
-
*
|
|
225
|
-
* because the observed blips are edge transients of seconds. */
|
|
228
|
+
/** The provider retry ladder: the first backoffs for a transport-class
|
|
229
|
+
* model-call failure. The final rung repeats while the run's loop lease has
|
|
230
|
+
* time, so these are pacing intervals rather than a three-attempt terminal
|
|
231
|
+
* budget. */
|
|
226
232
|
export const PROVIDER_RETRY_BACKOFFS_MS = [5_000, 15_000, 45_000] as const;
|
|
227
233
|
|
|
234
|
+
/** How often the model proxy writes an SSE comment while the provider is
|
|
235
|
+
* reasoning silently. The public container hop has cut an otherwise healthy
|
|
236
|
+
* stream at about thirty seconds of silence; a comment inside half that
|
|
237
|
+
* window keeps the transport alive without becoming a model event. The
|
|
238
|
+
* model call itself remains bounded by the run's lease. */
|
|
239
|
+
export const MODEL_STREAM_HEARTBEAT_MS = 15_000;
|
|
240
|
+
|
|
228
241
|
/** A hosted ship parent's deadline margin past the pipeline's wall clock
|
|
229
242
|
* (record 0060): the row's `state.hosting.until` is the hand-off time plus
|
|
230
243
|
* the instance's `caps.maxMinutes` plus this hour, absorbing the runner's own
|