@coreplane/switchboard 0.0.0 → 1.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +18 -1
  3. package/dist/assets/.dockerignore +27 -0
  4. package/dist/assets/.env.example +33 -0
  5. package/dist/assets/Dockerfile +111 -0
  6. package/dist/assets/config/config.example.yaml +359 -0
  7. package/dist/assets/deploy/bin/build-stamp.d.mts +15 -0
  8. package/dist/assets/deploy/bin/build-stamp.mjs +98 -0
  9. package/dist/assets/deploy/bin/cf-logs +32 -0
  10. package/dist/assets/deploy/cloudflare/package.json +29 -0
  11. package/dist/assets/deploy/cloudflare/preflight.mjs +243 -0
  12. package/dist/assets/deploy/cloudflare/tsconfig.json +18 -0
  13. package/dist/assets/deploy/cloudflare/worker.ts +382 -0
  14. package/dist/assets/deploy/cloudflare/wrangler.template.jsonc +67 -0
  15. package/dist/assets/deploy/cloudflare/write-build.d.mts +7 -0
  16. package/dist/assets/deploy/cloudflare/write-build.mjs +53 -0
  17. package/dist/assets/deploy/cloudflare-docs/package.json +18 -0
  18. package/dist/assets/deploy/cloudflare-docs/wrangler.template.jsonc +30 -0
  19. package/dist/assets/deploy/cloudflare-memory/package.json +25 -0
  20. package/dist/assets/deploy/cloudflare-memory/tsconfig.json +17 -0
  21. package/dist/assets/deploy/cloudflare-memory/worker.ts +2635 -0
  22. package/dist/assets/deploy/cloudflare-memory/wrangler.template.jsonc +50 -0
  23. package/dist/assets/deploy/cloudflare-resident/Dockerfile +91 -0
  24. package/dist/assets/deploy/cloudflare-resident/gc.ts +287 -0
  25. package/dist/assets/deploy/cloudflare-resident/node-async-hooks.d.ts +11 -0
  26. package/dist/assets/deploy/cloudflare-resident/package.json +29 -0
  27. package/dist/assets/deploy/cloudflare-resident/preflight.mjs +224 -0
  28. package/dist/assets/deploy/cloudflare-resident/tsconfig.json +19 -0
  29. package/dist/assets/deploy/cloudflare-resident/worker.ts +6637 -0
  30. package/dist/assets/deploy/cloudflare-resident/wrangler.template.jsonc +120 -0
  31. package/dist/assets/deploy/cloudflare-sandbox/Dockerfile +67 -0
  32. package/dist/assets/deploy/cloudflare-sandbox/docker-wrapper.sh +37 -0
  33. package/dist/assets/deploy/cloudflare-sandbox/package.json +26 -0
  34. package/dist/assets/deploy/cloudflare-sandbox/tsconfig.json +20 -0
  35. package/dist/assets/deploy/cloudflare-sandbox/worker.ts +410 -0
  36. package/dist/assets/deploy/cloudflare-sandbox/wrangler.template.jsonc +67 -0
  37. package/dist/assets/deploy/profile.example.json +13 -0
  38. package/dist/assets/deploy/secrets.manifest.json +108 -0
  39. package/dist/assets/docker-entrypoint.sh +15 -0
  40. package/dist/assets/package-lock.json +18407 -0
  41. package/dist/assets/package.json +104 -0
  42. package/dist/assets/project.json +219 -0
  43. package/dist/assets/source.json +5 -0
  44. package/dist/assets/src/core/authz/actor.ts +100 -0
  45. package/dist/assets/src/core/authz/authorize.ts +169 -0
  46. package/dist/assets/src/core/authz/grants.ts +347 -0
  47. package/dist/assets/src/core/authz/policy.ts +281 -0
  48. package/dist/assets/src/core/authz/resource.ts +147 -0
  49. package/dist/assets/src/core/authz/types.ts +164 -0
  50. package/dist/assets/src/core/drain.ts +54 -0
  51. package/dist/assets/src/core/ingressTokens.ts +64 -0
  52. package/dist/assets/src/core/memory/engine.ts +115 -0
  53. package/dist/assets/src/core/memory/scorer.ts +147 -0
  54. package/dist/assets/src/core/memory/types.ts +120 -0
  55. package/dist/assets/src/core/normalizeSpans.ts +299 -0
  56. package/dist/assets/src/core/prDescriptionTypes.ts +54 -0
  57. package/dist/assets/src/core/redact.ts +113 -0
  58. package/dist/assets/src/core/runEvents.ts +537 -0
  59. package/dist/assets/src/core/runFriction.ts +665 -0
  60. package/dist/assets/src/core/runLedger/decisions.ts +126 -0
  61. package/dist/assets/src/core/runLedger/types.ts +177 -0
  62. package/dist/assets/src/core/runRecord.ts +627 -0
  63. package/dist/assets/src/core/runShape.ts +61 -0
  64. package/dist/assets/src/core/schedules.ts +452 -0
  65. package/dist/assets/src/core/time/formatDuration.ts +61 -0
  66. package/dist/assets/src/core/trace/attrs.ts +203 -0
  67. package/dist/assets/src/core/trace/classify.ts +49 -0
  68. package/dist/assets/src/core/trace/clock.ts +6 -0
  69. package/dist/assets/src/core/trace/context.ts +9 -0
  70. package/dist/assets/src/core/trace/ids.ts +23 -0
  71. package/dist/assets/src/core/trace/partition.ts +235 -0
  72. package/dist/assets/src/core/trace/sinks.ts +68 -0
  73. package/dist/assets/src/core/trace/streamSpans.ts +163 -0
  74. package/dist/assets/src/core/trace/traceparent.ts +29 -0
  75. package/dist/assets/src/core/trace/tracer.ts +247 -0
  76. package/dist/assets/src/core/trace/types.ts +125 -0
  77. package/dist/assets/src/core/trace/workerTrace.ts +97 -0
  78. package/dist/assets/src/deploy/buildStamp.ts +93 -0
  79. package/dist/assets/src/deploy/liveGate.ts +203 -0
  80. package/dist/assets/src/deploy/profile.ts +162 -0
  81. package/dist/assets/src/deploy/restart.ts +393 -0
  82. package/dist/assets/src/effort.ts +17 -0
  83. package/dist/assets/src/execution/bashTimeout.ts +78 -0
  84. package/dist/assets/src/execution/bindingPurge.ts +43 -0
  85. package/dist/assets/src/execution/residentBackupTransfer.ts +50 -0
  86. package/dist/assets/src/execution/residentCleanliness.ts +95 -0
  87. package/dist/assets/src/execution/residentCredentials.ts +81 -0
  88. package/dist/assets/src/execution/residentDepCache.ts +321 -0
  89. package/dist/assets/src/execution/residentDepsStore.ts +326 -0
  90. package/dist/assets/src/execution/residentDetach.ts +48 -0
  91. package/dist/assets/src/execution/residentDisk.ts +107 -0
  92. package/dist/assets/src/execution/residentDiskBudget.ts +448 -0
  93. package/dist/assets/src/execution/residentExecWrap.ts +100 -0
  94. package/dist/assets/src/execution/residentHead.ts +85 -0
  95. package/dist/assets/src/execution/residentReadonly.ts +72 -0
  96. package/dist/assets/src/execution/residentRefresh.ts +429 -0
  97. package/dist/assets/src/execution/residentRestoreExtract.ts +130 -0
  98. package/dist/assets/src/execution/residentState.ts +47 -0
  99. package/dist/assets/src/execution/residentStepReport.ts +98 -0
  100. package/dist/assets/src/execution/residentStepTrace.ts +97 -0
  101. package/dist/assets/src/execution/residentSteps.ts +99 -0
  102. package/dist/assets/src/execution/residentText.ts +83 -0
  103. package/dist/assets/src/execution/residentTrace.ts +119 -0
  104. package/dist/assets/src/execution/sandboxEnv.ts +42 -0
  105. package/dist/assets/src/execution/sandboxErrors.ts +159 -0
  106. package/dist/assets/src/execution/sandboxKeepalive.ts +118 -0
  107. package/dist/assets/src/execution/shellQuote.ts +8 -0
  108. package/dist/assets/src/mcp/registry.ts +242 -0
  109. package/dist/assets/src/providers/types.ts +152 -0
  110. package/dist/assets/web/dist/.vite/manifest.json +176 -0
  111. package/dist/assets/web/dist/assets/AppShell-Bk2gbvet.js +1 -0
  112. package/dist/assets/web/dist/assets/CostsPage-CTZcMYYx.js +1 -0
  113. package/dist/assets/web/dist/assets/NotFoundPage-C-BuaSm8.js +1 -0
  114. package/dist/assets/web/dist/assets/ResidentDetailPage-D3shEnzl.js +1 -0
  115. package/dist/assets/web/dist/assets/ResidentsIndexPage-DWIubQ05.js +1 -0
  116. package/dist/assets/web/dist/assets/RunRoutePage-BMjuE-oX.js +126 -0
  117. package/dist/assets/web/dist/assets/RunRoutePage-XVFj0XDc.css +1 -0
  118. package/dist/assets/web/dist/assets/RunsIndexPage-C3_jYIo0.js +1 -0
  119. package/dist/assets/web/dist/assets/RunsTabs-C4krAL9o.js +1 -0
  120. package/dist/assets/web/dist/assets/ScheduledPage-g1W58mtN.js +1 -0
  121. package/dist/assets/web/dist/assets/StatusDot-DcPRw3zu.js +1 -0
  122. package/dist/assets/web/dist/assets/Tooltip-DJUkMYjo.js +1 -0
  123. package/dist/assets/web/dist/assets/favicon-DL1rdWJt.js +1 -0
  124. package/dist/assets/web/dist/assets/localIso-L06jV29p.js +1 -0
  125. package/dist/assets/web/dist/assets/main-BsBGUyMH.css +2 -0
  126. package/dist/assets/web/dist/assets/main-CyM5f4JC.js +28 -0
  127. package/dist/assets/web/dist/assets/residentDiskBudget-BMBKlYRH.js +1 -0
  128. package/dist/assets/web/dist/assets/seed-BglCRKLA.js +6 -0
  129. package/dist/assets/web/dist/assets/wallClock-Ckv3sKoR.js +1 -0
  130. package/dist/cli.js +34494 -0
  131. package/package.json +43 -10
@@ -0,0 +1,243 @@
1
+ #!/usr/bin/env node
2
+ // Deploy preflight for the bot Worker (docs/reference/specs/slack-channel.md item 8).
3
+ //
4
+ // `wrangler deploy` rolls the bot container. Cloudflare's rollout sends SIGTERM;
5
+ // since the run ledger's handoff (docs/reference/specs/run-history.md item 39) the bot
6
+ // hands every resumable run to the next generation and exits within seconds,
7
+ // and the next generation continues the runs under their own cards — so a
8
+ // deploy no longer waits on runs, and this preflight no longer refuses for
9
+ // them. `npm run deploy` runs it first and refuses only while
10
+ // - the container application is not in a settled state (a rollout is still
11
+ // provisioning/updating — `wrangler containers list --json`): a second
12
+ // rollout on top of one in progress replaces the instance the first put
13
+ // into its graceful drain and kills whatever it was running (two deploys
14
+ // 90 s apart once killed a review at 153 s).
15
+ // It WARNS (never refuses) when the bot reports runs in flight (`inFlight > 0`
16
+ // — they hand off) or is already draining (`draining: true` — its resumable
17
+ // runs were handed off; a ship pipeline still in flight would be killed), and
18
+ // when /healthz says the reconnect catch-up is failing or the bot token lacks
19
+ // required scopes (`catchUp.error`, `catchUp.missingScopes`).
20
+ //
21
+ // Fail closed: unreachable bot, a body without the JSON shape (a bot whose
22
+ // /healthz answers a bare `ok` is not one this preflight can read), a wrangler failure, or an app
23
+ // not in the listing all refuse — a deploy never proceeds blind.
24
+ // `SWITCHBOARD_DEPLOY_FORCE=1` (or `--force` when run directly) is the
25
+ // explicit, warned bypass.
26
+ //
27
+ // Dependency-free Node (fetch + child_process). `decide()` is pure and
28
+ // unit-tested (preflight.test.mjs); `main()` only does I/O around it.
29
+ import { execFile } from "node:child_process";
30
+ import { dirname } from "node:path";
31
+ import { fileURLToPath, pathToFileURL } from "node:url";
32
+
33
+ /** The env var naming this Worker's origin. `deploy all` sets it from the deployment profile; the
34
+ * preflight has no address of its own (deploy/profile.json is the only place the fleet's hostnames live). */
35
+ export const BASE_URL_ENV = "SWITCHBOARD_BASE_URL";
36
+ /** The Containers application `wrangler deploy` creates for `SwitchboardServer` in wrangler.jsonc. */
37
+ export const APP_NAME = "switchboard-switchboardserver";
38
+ /** Application states in which no rollout is in progress. Anything else —
39
+ * `provisioning`, `updating`, or a state this script does not know — refuses. */
40
+ const SETTLED_APP_STATES = new Set(["active", "ready"]);
41
+
42
+ const HOW_TO_FORCE =
43
+ "to deploy anyway (over a rollout in progress, or blind when the bot cannot be consulted): `SWITCHBOARD_DEPLOY_FORCE=1 npm run deploy` (`node preflight.mjs --force` checks alone)";
44
+
45
+ /** GET /healthz. Never throws: `{ok:true,payload}` (parsed JSON, or the raw text when not JSON) or `{ok:false,error}`. */
46
+ export async function fetchHealth(baseUrl, { timeoutMs = 20_000 } = {}) {
47
+ const url = new URL("/healthz", baseUrl).toString();
48
+ try {
49
+ const res = await fetch(url, { signal: AbortSignal.timeout(timeoutMs) });
50
+ const text = await res.text();
51
+ if (!res.ok) return { ok: false, error: `GET ${url} → HTTP ${res.status}: ${text.slice(0, 200)}` };
52
+ try {
53
+ return { ok: true, payload: JSON.parse(text) };
54
+ } catch {
55
+ return { ok: true, payload: text };
56
+ }
57
+ } catch (err) {
58
+ return { ok: false, error: `GET ${url} failed: ${err instanceof Error ? err.message : String(err)}` };
59
+ }
60
+ }
61
+
62
+ /**
63
+ * The words to keep from a failed wrangler command. wrangler prints its
64
+ * errors on STDOUT (`✘ [ERROR] A request to the Cloudflare API … failed.
65
+ * Authentication error [code: 10000]`), so a message built from stderr alone
66
+ * says "Command failed: npx wrangler containers list --json" and nothing else
67
+ * — which is exactly what the first CI release deploy showed while its token
68
+ * lacked the Containers scope. Prefer wrangler's own error lines; else the
69
+ * last non-empty lines of both streams; else the exit description. Pure.
70
+ */
71
+ /** ANSI colour sequences (ESC `[` … `m`), built from the code point so the regex literal carries no control character. */
72
+ const ANSI_SEQUENCE = new RegExp(`${String.fromCharCode(27)}\\[[0-9;]*m`, "g");
73
+
74
+ export function wranglerFailureText(err, stdout, stderr, { maxLines = 3 } = {}) {
75
+ const lines = `${String(stdout ?? "")}\n${String(stderr ?? "")}`
76
+ .split("\n")
77
+ .map((l) => l.replace(ANSI_SEQUENCE, "").trim())
78
+ .filter((l) => l.length > 0 && !/^npm (ERR|WARN|warn)/i.test(l));
79
+ const errorLines = lines.filter((l) =>
80
+ /\[ERROR\]|✘|error|Authentication|Unauthorized|not authorized|permission/i.test(l),
81
+ );
82
+ const picked = (errorLines.length > 0 ? errorLines : lines).slice(-maxLines);
83
+ const exit =
84
+ err && typeof err.code === "number" ? `exit ${err.code}` : err && err.killed ? "killed (timeout?)" : "failed";
85
+ return picked.length > 0 ? `${exit}: ${picked.join(" | ")}` : `${exit}, no output`;
86
+ }
87
+
88
+ /** `wrangler containers list --json`, run from this directory so wrangler.jsonc
89
+ * selects the account. Never throws: `{ok:true,payload}` or `{ok:false,error}`. */
90
+ export function listContainerApps({ cwd = dirname(fileURLToPath(import.meta.url)), timeoutMs = 60_000 } = {}) {
91
+ return new Promise((resolve) => {
92
+ execFile(
93
+ "npx",
94
+ ["wrangler", "containers", "list", "--json"],
95
+ { cwd, timeout: timeoutMs, maxBuffer: 4 * 1024 * 1024 },
96
+ (err, stdout, stderr) => {
97
+ if (err)
98
+ return resolve({
99
+ ok: false,
100
+ error: `wrangler containers list failed (${wranglerFailureText(err, stdout, stderr)}) — the credential may lack the Containers scope`,
101
+ });
102
+ // wrangler prints its banner before the JSON; the payload starts at the first `[`.
103
+ const start = String(stdout).indexOf("[");
104
+ if (start < 0)
105
+ return resolve({
106
+ ok: false,
107
+ error: `wrangler containers list: no JSON array in output: ${String(stdout).slice(0, 200)}`,
108
+ });
109
+ try {
110
+ resolve({ ok: true, payload: JSON.parse(String(stdout).slice(start)) });
111
+ } catch (parseErr) {
112
+ resolve({
113
+ ok: false,
114
+ error: `wrangler containers list: unparsable JSON: ${parseErr instanceof Error ? parseErr.message : String(parseErr)}`,
115
+ });
116
+ }
117
+ },
118
+ );
119
+ });
120
+ }
121
+
122
+ /**
123
+ * Warnings (never refusals) from the reconnect catch-up's status on /healthz
124
+ * (docs/reference/specs/slack-channel.md item 7): a scan that could not run at all
125
+ * (`catchUp.error`, e.g. `missing_scope`) or a bot token missing required
126
+ * scopes. Pure; an older payload without `catchUp` says nothing.
127
+ * @returns {string[]}
128
+ */
129
+ export function catchUpWarnings(payload) {
130
+ const c = payload && typeof payload === "object" ? payload.catchUp : undefined;
131
+ if (!c || typeof c !== "object") return [];
132
+ const out = [];
133
+ if (typeof c.error === "string" && c.error) {
134
+ out.push(
135
+ `reconnect catch-up is NOT running (last attempt${c.lastRunAt ? ` ${c.lastRunAt}` : ""}): ${c.error} — mentions posted during a rollover are being dropped`,
136
+ );
137
+ }
138
+ if (Array.isArray(c.missingScopes) && c.missingScopes.length > 0) {
139
+ out.push(
140
+ `bot token is missing required Slack scopes: ${c.missingScopes.join(", ")} — reinstall the app with them (docs/tutorials/run-it-locally.md → Connect it to Slack)`,
141
+ );
142
+ }
143
+ return out;
144
+ }
145
+
146
+ /**
147
+ * The decision, pure. `health` is the result of `fetchHealth`, `apps` of `listContainerApps`.
148
+ * `warnings` never block: this deploy may be the fix for what they name.
149
+ * @returns {{ allow: boolean, forced: boolean, problems: string[], warnings: string[], message: string }}
150
+ */
151
+ export function decide({ health, apps }, { force = false } = {}) {
152
+ const problems = [];
153
+ const warnings = health?.ok === true ? catchUpWarnings(health.payload) : [];
154
+
155
+ if (!health || health.ok !== true) {
156
+ problems.push(`bot not consulted: ${health?.error ?? "unknown error"}`);
157
+ } else {
158
+ const p = health.payload;
159
+ if (!p || typeof p !== "object") {
160
+ problems.push(
161
+ `bot answered /healthz without JSON (${JSON.stringify(p).slice(0, 40)}) — its /healthz is not the JSON this preflight reads; deploy once with SWITCHBOARD_DEPLOY_FORCE=1`,
162
+ );
163
+ } else {
164
+ if (!Number.isInteger(p.inFlight) || p.inFlight < 0) {
165
+ problems.push(`bot reports an impossible inFlight=${JSON.stringify(p.inFlight)} (counter bug or old Worker)`);
166
+ } else if (p.inFlight > 0) {
167
+ // Not a refusal since the handoff (run-history item 39): SIGTERM hands
168
+ // every resumable run to the next generation, which continues it.
169
+ warnings.push(
170
+ `${p.inFlight} run(s) in flight — handed to the next generation on SIGTERM (run-history item 39); they continue there under their own cards`,
171
+ );
172
+ }
173
+ if (p.draining === true) {
174
+ warnings.push(
175
+ "bot is already draining from a previous deploy — its resumable runs are handed off; a ship pipeline still in flight would be killed when this rollout replaces the draining instance",
176
+ );
177
+ }
178
+ }
179
+ }
180
+
181
+ if (!apps || apps.ok !== true) {
182
+ problems.push(`container application not consulted: ${apps?.error ?? "unknown error"}`);
183
+ } else {
184
+ const app = Array.isArray(apps.payload) ? apps.payload.find((a) => a?.name === APP_NAME) : undefined;
185
+ if (!app) {
186
+ problems.push(
187
+ `container application ${APP_NAME} not in \`wrangler containers list\` (wrong account, or renamed class?)`,
188
+ );
189
+ } else if (!SETTLED_APP_STATES.has(app.state)) {
190
+ problems.push(`container rollout in progress: state=${app.state} — wait until it is active`);
191
+ }
192
+ }
193
+
194
+ const warningText =
195
+ warnings.length > 0 ? `\n WARNING (not blocking):\n${warnings.map((w) => ` - ${w}`).join("\n")}` : "";
196
+ if (problems.length === 0) {
197
+ return {
198
+ allow: true,
199
+ forced: false,
200
+ problems,
201
+ warnings,
202
+ message: `preflight ok: container application settled${warningText}`,
203
+ };
204
+ }
205
+ const detail = problems.map((p) => ` - ${p}`).join("\n");
206
+ if (force) {
207
+ return {
208
+ allow: true,
209
+ forced: true,
210
+ problems,
211
+ warnings,
212
+ message: `preflight WARNING: deploying by force despite —\n${detail}\n a rollout landing on one in progress can disrupt it; in-flight runs hand off regardless (run-history item 39)${warningText}`,
213
+ };
214
+ }
215
+ return {
216
+ allow: false,
217
+ forced: false,
218
+ problems,
219
+ warnings,
220
+ message: `preflight REFUSED: a Worker deploy rolls the bot container —\n${detail}\n wait and retry; ${HOW_TO_FORCE}${warningText}`,
221
+ };
222
+ }
223
+
224
+ export async function main(argv = process.argv.slice(2), env = process.env) {
225
+ const force = argv.includes("--force") || env.SWITCHBOARD_DEPLOY_FORCE === "1";
226
+ const baseUrl = env[BASE_URL_ENV];
227
+ if (!baseUrl) {
228
+ console.error(
229
+ `[bot-preflight] ${BASE_URL_ENV} is not set — the bot's origin comes from the deployment profile; deploy through \`npm run cli -- deploy all\` (it sets it), or set it to https://<bot hostname> to run this alone`,
230
+ );
231
+ return 2;
232
+ }
233
+ const [health, apps] = await Promise.all([fetchHealth(baseUrl), listContainerApps()]);
234
+ const d = decide({ health, apps }, { force });
235
+ (d.allow && !d.forced && d.warnings.length === 0 ? console.log : console.error)(`[bot-preflight] ${d.message}`);
236
+ return d.allow ? 0 : 1;
237
+ }
238
+
239
+ // Run only when executed directly (`node preflight.mjs`), not when imported by tests.
240
+ // pathToFileURL, not `file://${argv[1]}`, so the guard also holds on Windows paths.
241
+ if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) {
242
+ process.exitCode = await main();
243
+ }
@@ -0,0 +1,18 @@
1
+ {
2
+ // Scoped config so `npm run typecheck` here checks worker.ts AND the shared
3
+ // src/core/schedules.ts it imports by relative path (the schedule registry +
4
+ // the pure firing helpers). Strict, like the bot — the shared module is
5
+ // written for strict narrowing.
6
+ "compilerOptions": {
7
+ "target": "es2022",
8
+ "module": "es2022",
9
+ "moduleResolution": "bundler",
10
+ "lib": ["es2022"],
11
+ "types": ["@cloudflare/workers-types"],
12
+ "strict": true,
13
+ "noEmit": true,
14
+ "skipLibCheck": true,
15
+ "allowImportingTsExtensions": true
16
+ },
17
+ "include": ["worker.ts"]
18
+ }
@@ -0,0 +1,382 @@
1
+ // Cloudflare Containers shim: runs the unchanged Switchboard image as a single
2
+ // always-on container instance — the shape of any long-lived server on
3
+ // Cloudflare Containers (singleton DO, cron keep-alive;
4
+ // docs/decisions/0016-long-lived-process-not-serverless.md).
5
+ //
6
+ // The Worker exists to (re)start the container, run health checks, restart the
7
+ // container on request (`POST /admin/restart` — `deploy restart`), and fire
8
+ // the scheduled jobs — all Slack traffic is the container's own outbound Socket
9
+ // Mode websocket, so nothing user-facing flows through here. Scheduled jobs are
10
+ // NOT special: a `run` schedule is POSTed to the bot's generic /ingress as the
11
+ // `cron` identity, so it becomes an ordinary run.
12
+ import { Container, getContainer } from "@cloudflare/containers";
13
+ import { parseHealthz } from "../../src/deploy/liveGate.ts";
14
+ import {
15
+ authenticateRestart,
16
+ decideRestart,
17
+ parseRestartAuthorization,
18
+ parseRestartRequest,
19
+ RESTART_AUTHORIZE_PATH,
20
+ RESTART_SUBJECT_HEADER,
21
+ restartResponse,
22
+ stripRestartSubject,
23
+ type RestartAuth,
24
+ type RestartOutcome,
25
+ } from "../../src/deploy/restart.ts";
26
+ import {
27
+ interpretIngressResponse,
28
+ isRunSchedule,
29
+ planScheduledFiring,
30
+ recordFiring as postFiring,
31
+ scheduleForCron,
32
+ type ScheduleFiring,
33
+ } from "../../src/core/schedules.ts";
34
+ import { systemClock } from "../../src/core/trace/clock.ts";
35
+ import { createTracer } from "../../src/core/trace/tracer.ts";
36
+ import { shimRoute, stripTraceContext, withTraceContext, workerLogSink } from "../../src/core/trace/workerTrace.ts";
37
+
38
+ // The Worker's own spans (docs/reference/specs/tracing.md item 22): one `bot-shim.fetch`
39
+ // root per routed request and one `cron.<schedule>` root per fired schedule,
40
+ // on a `slow` log sink whose filter drops the line an unauthenticated refusal
41
+ // would leave. The public edge never adopts a caller's trace context.
42
+ const tracer = createTracer({ clock: systemClock });
43
+ const traceSinks = [workerLogSink((line) => console.log(line))];
44
+
45
+ interface Env {
46
+ SWITCHBOARD: DurableObjectNamespace<SwitchboardServer>;
47
+ // secrets (wrangler secret put ...)
48
+ SLACK_BOT_TOKEN: string;
49
+ SLACK_APP_TOKEN: string;
50
+ ANTHROPIC_API_KEY: string;
51
+ OPENAI_API_KEY?: string;
52
+ E2B_API_KEY?: string;
53
+ SANDBOX_TOKEN?: string; // cloudflare execution: bearer for the sandbox Worker
54
+ RESIDENT_OPERATOR_TOKEN?: string; // resident repos: operator bearer for the resident Worker
55
+ RESIDENT_ADMIN_TOKEN?: string; // resident repos: admin bearer for `repo onboard/offboard/...` chat commands
56
+ GH_TOKEN?: string; // fallback when no GitHub App is configured
57
+ GITHUB_APP_ID?: string;
58
+ GITHUB_APP_INSTALLATION_ID?: string;
59
+ GITHUB_APP_PRIVATE_KEY?: string;
60
+ PUBLIC_BASE_URL?: string; // live-view: base for /runs/<id>?t=… links on the status card
61
+ ACCESS_TEAM_DOMAIN?: string; // live-view SSO gate: Cloudflare Access team domain (JWKS + iss)
62
+ ACCESS_AUD?: string; // live-view SSO gate: Cloudflare Access application AUD tag
63
+ DASHBOARD_TOKEN?: string; // dashboard auth `token` strategy: the bearer (the default env name; config may name another)
64
+ SWITCHBOARD_INGRESS_TOKENS?: string; // enables HTTP /ingress + MCP /mcp (JSON token→identity map); the `cron` entry is what scheduled runs present; an entry whose `http:<subject>` actor holds `deploy:write` in the bot's config may POST /admin/restart
65
+ BRAVE_SEARCH_API_KEY?: string; // web_search backend (Brave); web_fetch works without it
66
+ CF_ANALYTICS_TOKEN?: string; // costs dash: Cloudflare API token, Account Analytics:Read only
67
+ ANTHROPIC_ADMIN_KEY?: string; // costs dash (optional): Anthropic Admin API key for the LLM cost report
68
+ MEMORY_TOKEN?: string; // durable memory + friction ledger + schedule firings + MCP registry: bearer for the state Worker
69
+ MCP_CREDENTIAL_KEY?: string; // MCP registry: the bot-only key that seals server credentials before they reach the McpDO
70
+ STATE_WORKER_URL?: string; // var: the state Worker's base URL — where this shim records each scheduled firing
71
+ }
72
+
73
+ /** Every secret/var the Worker forwards into the container. Optional entries
74
+ * are forwarded only when set, so the bot sees "not configured" as absence. */
75
+ const FORWARDED_OPTIONAL = [
76
+ "OPENAI_API_KEY",
77
+ "E2B_API_KEY",
78
+ "SANDBOX_TOKEN",
79
+ "RESIDENT_OPERATOR_TOKEN",
80
+ "RESIDENT_ADMIN_TOKEN",
81
+ "GH_TOKEN",
82
+ "GITHUB_APP_ID",
83
+ "GITHUB_APP_INSTALLATION_ID",
84
+ "GITHUB_APP_PRIVATE_KEY",
85
+ "PUBLIC_BASE_URL",
86
+ "ACCESS_TEAM_DOMAIN",
87
+ "ACCESS_AUD",
88
+ "DASHBOARD_TOKEN",
89
+ "CF_ANALYTICS_TOKEN",
90
+ "ANTHROPIC_ADMIN_KEY",
91
+ "SWITCHBOARD_INGRESS_TOKENS",
92
+ "BRAVE_SEARCH_API_KEY",
93
+ "MEMORY_TOKEN",
94
+ "MCP_CREDENTIAL_KEY",
95
+ "STATE_WORKER_URL",
96
+ ] as const satisfies readonly (keyof Env)[];
97
+
98
+ /** The container's environment, computed from the Worker env AT START TIME.
99
+ * A running container keeps the env it started with, whatever `wrangler
100
+ * secret put` has since changed on the Worker — so this is read on every
101
+ * (re)start rather than once in the constructor: after `deploy restart`
102
+ * stops the container, the next start carries the current secrets. */
103
+ function containerEnv(env: Env): Record<string, string> {
104
+ const vars: Record<string, string> = {
105
+ // The image carries no config: the bot reads the `base` document `deploy config`
106
+ // pushed to the state Worker (src/configDocument.ts), reached through
107
+ // STATE_WORKER_URL + MEMORY_TOKEN forwarded below.
108
+ SWITCHBOARD_CONFIG: "state://base",
109
+ SLACK_BOT_TOKEN: env.SLACK_BOT_TOKEN,
110
+ SLACK_APP_TOKEN: env.SLACK_APP_TOKEN,
111
+ ANTHROPIC_API_KEY: env.ANTHROPIC_API_KEY,
112
+ };
113
+ for (const name of FORWARDED_OPTIONAL) {
114
+ const value = env[name];
115
+ if (value) vars[name] = value;
116
+ }
117
+ return vars;
118
+ }
119
+
120
+ const INSTANCE = "singleton";
121
+ const INTERNAL = "https://switchboard-keepalive.internal";
122
+
123
+ export class SwitchboardServer extends Container<Env> {
124
+ defaultPort = 8080; // the bot's health endpoint (PORT=8080 in the image)
125
+ // Never let this scale to zero: the Slack websocket must stay connected and
126
+ // Slack does not redeliver missed Socket Mode events. The keep-alive cron
127
+ // pings /healthz every MINUTE (wrangler.jsonc `* * * * *`, deliberate: it
128
+ // bounds the Slack deaf window after a stop/rollover — the sooner a touch
129
+ // restarts the container, the sooner the reconnect catch-up can run).
130
+ sleepAfter = "2h";
131
+
132
+ /** Start the container if it is not running, with the env computed now.
133
+ * Default port-ready timeout is 20s; first boot (npm-less image, but cold
134
+ * pull + Slack connect) can exceed it. Already running → a no-op (the
135
+ * start options are not applied to a live container). */
136
+ private startBot(): Promise<void> {
137
+ return this.startAndWaitForPorts(
138
+ this.defaultPort,
139
+ { portReadyTimeoutMS: 120_000 },
140
+ { envVars: containerEnv(this.env) },
141
+ );
142
+ }
143
+
144
+ override async fetch(request: Request): Promise<Response> {
145
+ await this.startBot();
146
+ return super.fetch(request);
147
+ }
148
+
149
+ /**
150
+ * `deploy restart`: stop the container WITHOUT an image build so it comes
151
+ * back on the Worker's current secrets. Cloudflare's idiom — `stop()` sends
152
+ * SIGTERM, the bot's graceful drain (src/index.ts) finishes in-flight runs
153
+ * and exits, and the NEXT request through `fetch` starts the container again
154
+ * (`startBot`, env computed then). The keep-alive cron GETs /healthz every
155
+ * minute and the CLI's live gate polls it every 15 s, so the next request is
156
+ * never more than seconds away. Refuses (the deploy preflight's rules) while
157
+ * runs are in flight or a drain is already under way unless `force`.
158
+ */
159
+ async restart(opts: { force: boolean }): Promise<RestartOutcome> {
160
+ if (!this.ctx.container?.running) return { kind: "not-running" };
161
+ return this.restartRunning(opts);
162
+ }
163
+
164
+ /**
165
+ * `deploy restart`, authorized: the Worker knows WHO the bearer is (the token
166
+ * map); WHETHER that identity may restart is the bot's config (`grants` —
167
+ * authorization.md item 9), which only the container holds. So ask it —
168
+ * `POST /admin/restart/authorize` with the authenticated subject in
169
+ * `RESTART_SUBJECT_HEADER` (never the bearer: the container may still hold the
170
+ * token map from before a rotation) — and stop only on a 200. A container that
171
+ * is not running is started first (the bot must answer): if the bearer is
172
+ * allowed, that start already put the current env live, so nothing is
173
+ * stopped and the outcome is `not-running`, exactly as before; if not, the
174
+ * refusal is relayed and the started container simply keeps serving.
175
+ */
176
+ async restartAuthorized(
177
+ subject: string,
178
+ opts: { force: boolean },
179
+ ): Promise<{ auth: RestartAuth; outcome?: RestartOutcome }> {
180
+ const wasRunning = this.ctx.container?.running === true;
181
+ await this.startBot();
182
+ const answer = await this.containerFetch(
183
+ new Request(`${INTERNAL}${RESTART_AUTHORIZE_PATH}`, {
184
+ method: "POST",
185
+ headers: { [RESTART_SUBJECT_HEADER]: subject },
186
+ }),
187
+ this.defaultPort,
188
+ );
189
+ const auth = parseRestartAuthorization(answer.status, await answer.text().catch(() => ""));
190
+ if (!auth.ok) return { auth };
191
+ if (!wasRunning) return { auth, outcome: { kind: "not-running" } };
192
+ return { auth, outcome: await this.restartRunning(opts) };
193
+ }
194
+
195
+ /** The stop itself, for a running container: the preflight's refusal rules over `/healthz`, then SIGTERM. */
196
+ private async restartRunning(opts: { force: boolean }): Promise<RestartOutcome> {
197
+ const health = await this.containerFetch(new Request(`${INTERNAL}/healthz`), this.defaultPort);
198
+ const body = parseHealthz(await health.text().catch(() => ""));
199
+ const verdict = decideRestart(body, opts);
200
+ if (!verdict.allow) return { kind: "refused", problems: verdict.problems };
201
+ const inFlight = typeof body?.inFlight === "number" ? body.inFlight : 0;
202
+ const previousStartedAt = typeof body?.startedAt === "string" ? body.startedAt : undefined;
203
+ console.log(
204
+ `[restart] SIGTERM → container (started ${previousStartedAt ?? "unknown"}, ${inFlight} in flight${verdict.forced ? ", FORCED" : ""})`,
205
+ );
206
+ await this.stop();
207
+ return { kind: "stopping", forced: verdict.forced, inFlight, previousStartedAt };
208
+ }
209
+
210
+ override onStop(params: { exitCode: number; reason: string }): void {
211
+ // The next fetch (cron keep-alive within a minute, or the CLI's poll) starts it again with the current env.
212
+ console.log(
213
+ `[restart] container stopped (exit ${params.exitCode}, ${params.reason}) — restarts with the current env on the next request`,
214
+ );
215
+ }
216
+ }
217
+
218
+ /** `POST /admin/restart` — the operator surface behind `deploy restart`
219
+ * (src/deploy/restart.ts documents the authorization choice: a
220
+ * SWITCHBOARD_INGRESS_TOKENS bearer whose `http:<subject>` actor holds
221
+ * `deploy:write` in the bot's config). The Worker authenticates the bearer
222
+ * against the map it holds — an unknown bearer never touches the container —
223
+ * and the Container DO asks the bot for the grant before stopping anything.
224
+ * Body `{ "force": true }` bypasses the in-flight/draining refusal. */
225
+ async function handleAdminRestart(request: Request, env: Env): Promise<Response> {
226
+ const json = (status: number, body: Record<string, unknown>) =>
227
+ new Response(JSON.stringify(body), { status, headers: { "content-type": "application/json" } });
228
+ if (request.method !== "POST") return json(405, { ok: false, error: "method not allowed: POST /admin/restart" });
229
+ const authorization = request.headers.get("authorization") ?? undefined;
230
+ const authn = authenticateRestart(authorization, env.SWITCHBOARD_INGRESS_TOKENS);
231
+ if (!authn.ok) {
232
+ console.warn(`[restart] ${authn.status} — ${authn.reason}`);
233
+ return json(authn.status, { ok: false, error: authn.reason });
234
+ }
235
+ const parsed = parseRestartRequest(await request.text().catch(() => ""));
236
+ if (!parsed.ok) return json(400, { ok: false, error: parsed.reason });
237
+ let auth: RestartAuth;
238
+ let outcome: RestartOutcome | undefined;
239
+ try {
240
+ ({ auth, outcome } = await getContainer(env.SWITCHBOARD, INSTANCE).restartAuthorized(authn.identity.subject, {
241
+ force: parsed.force,
242
+ }));
243
+ } catch (err) {
244
+ // The container's /healthz probe or the DO call threw (container mid-transition,
245
+ // port not answering): fail closed in the route's own JSON shape so the CLI reads
246
+ // a reason instead of the platform's HTML 500. Nothing was stopped.
247
+ const reason = err instanceof Error ? err.message : String(err);
248
+ console.error(`[restart] ${authn.identity.subject} → error before stop: ${reason}`);
249
+ return json(500, { ok: false, error: `restart failed before stopping anything: ${reason}` });
250
+ }
251
+ if (!auth.ok || outcome === undefined) {
252
+ const refusal = auth.ok ? { status: 503 as const, reason: "restart disabled: no outcome" } : auth;
253
+ console.warn(`[restart] ${refusal.status} — ${refusal.reason}`);
254
+ return json(refusal.status, { ok: false, error: refusal.reason });
255
+ }
256
+ console.log(
257
+ `[restart] ${auth.subject} → ${outcome.kind}${outcome.kind === "refused" ? `: ${outcome.problems.join("; ")}` : ""}`,
258
+ );
259
+ const res = restartResponse(outcome);
260
+ return json(res.status, res.body);
261
+ }
262
+
263
+ /** Record a firing on the state Worker's ScheduleDO (the /runs Scheduled panel
264
+ * reads it). Best-effort: a failure here is a log line — the run itself (if
265
+ * any) already happened and is its own record. */
266
+ async function recordFiring(env: Env, firing: ScheduleFiring): Promise<void> {
267
+ const res = await postFiring({ url: env.STATE_WORKER_URL, token: env.MEMORY_TOKEN }, firing);
268
+ if (!res.ok) console.error(`[schedule] ${firing.schedule}: recording the firing failed — ${res.reason}`);
269
+ }
270
+
271
+ export default {
272
+ async fetch(request: Request, env: Env): Promise<Response> {
273
+ const pathname = new URL(request.url).pathname;
274
+ // The public edge (docs/reference/specs/tracing.md item 22): whatever trace context the
275
+ // caller sent is stripped, and what the container sees carries this
276
+ // Worker's own root. A static asset or the live view's SSE stream gets no
277
+ // root; a refusal's line is dropped by the sink's filter.
278
+ // The restart-subject header is the Worker's own word to the container (the
279
+ // authorize call below); a caller cannot be allowed to speak it.
280
+ const inbound = stripRestartSubject(stripTraceContext(request));
281
+ const route = shimRoute(pathname);
282
+ if (route === undefined) return getContainer(env.SWITCHBOARD, INSTANCE).fetch(inbound);
283
+ const root = tracer.start("bot-shim.fetch", { sinks: traceSinks, attrs: { route } });
284
+ try {
285
+ const forwarded = withTraceContext(inbound, root);
286
+ // The one route the Worker answers itself; everything else is the container's.
287
+ const res =
288
+ pathname === "/admin/restart"
289
+ ? await handleAdminRestart(forwarded, env)
290
+ : await getContainer(env.SWITCHBOARD, INSTANCE).fetch(forwarded);
291
+ root.end(res.status >= 500 ? "error" : "ok", { httpStatus: res.status });
292
+ return res;
293
+ } catch (err) {
294
+ root.fail(err);
295
+ root.end("error");
296
+ throw err;
297
+ }
298
+ },
299
+
300
+ // Every cron trigger in wrangler.jsonc is a `bot` schedule in the registry
301
+ // (src/core/schedules.ts — a unit test keeps the two equal). `healthz` (the
302
+ // internal keep-alive) touches /healthz: any touch starts the container if
303
+ // stopped and renews the activity timeout, which is also what revives it after
304
+ // platform maintenance; internal, so nothing is recorded. A `run` schedule
305
+ // POSTs the bot's generic /ingress as the `cron` identity (its bearer is the
306
+ // `cron` entry of SWITCHBOARD_INGRESS_TOKENS — no extra secret) with the
307
+ // schedule's command text; the bot dispatches it as a normal run and answers
308
+ // with the run's id + status, which is recorded on the state Worker for the
309
+ // /runs Scheduled panel. Fail-closed: no `cron` token → nothing is sent and the
310
+ // firing is recorded as `misconfigured`; an ingress error (bot down, unknown
311
+ // identity) is recorded as `ingress-error`.
312
+ async scheduled(controller: ScheduledController, env: Env, ctx: ExecutionContext): Promise<void> {
313
+ const schedule = scheduleForCron(controller.cron, "bot");
314
+ if (!schedule) {
315
+ console.error(
316
+ `[schedule] cron "${controller.cron}" is not a bot schedule in the registry — nothing fired (wrangler.jsonc and src/core/schedules.ts have drifted)`,
317
+ );
318
+ return;
319
+ }
320
+ if (schedule.action.type === "healthz") {
321
+ const res = await getContainer(env.SWITCHBOARD, INSTANCE).fetch(new Request(`${INTERNAL}/healthz`));
322
+ if (!res.ok) console.error(`switchboard health check failed: ${res.status}`);
323
+ return;
324
+ }
325
+ if (!isRunSchedule(schedule)) {
326
+ console.error(
327
+ `[schedule] ${schedule.name}: action "${schedule.action.type}" is not something the bot shim fires — nothing fired (the registry entry names the wrong worker)`,
328
+ );
329
+ return;
330
+ }
331
+
332
+ const firedAt = controller.scheduledTime || systemClock();
333
+ const plan = planScheduledFiring(schedule, env.SWITCHBOARD_INGRESS_TOKENS, firedAt);
334
+ if (!plan.ok) {
335
+ console.error(`[schedule] ${schedule.name}: not armed — ${plan.reason}; nothing ran`);
336
+ ctx.waitUntil(
337
+ recordFiring(env, { schedule: schedule.name, firedAt, outcome: "misconfigured", detail: plan.reason }),
338
+ );
339
+ return;
340
+ }
341
+ // The firing is one `cron.<schedule>` root (docs/reference/specs/tracing.md item 22):
342
+ // the ingress request carries it, and the recorded firing names its trace.
343
+ const root = tracer.start(`cron.${schedule.name}`, { sinks: traceSinks, startedAt: firedAt });
344
+ let firing: ScheduleFiring;
345
+ try {
346
+ const res = await getContainer(env.SWITCHBOARD, INSTANCE).fetch(
347
+ withTraceContext(
348
+ new Request(`${INTERNAL}/ingress`, {
349
+ method: "POST",
350
+ headers: { "content-type": "application/json", authorization: `Bearer ${plan.token}` },
351
+ body: JSON.stringify(plan.body),
352
+ }),
353
+ root,
354
+ ),
355
+ );
356
+ firing = {
357
+ ...interpretIngressResponse(schedule, firedAt, res.status, await res.text().catch(() => "")),
358
+ traceId: root.traceId,
359
+ };
360
+ } catch (err) {
361
+ firing = {
362
+ schedule: schedule.name,
363
+ firedAt,
364
+ outcome: "ingress-error",
365
+ detail: `fetch failed: ${err instanceof Error ? err.message : String(err)}`.slice(0, 300),
366
+ traceId: root.traceId,
367
+ };
368
+ }
369
+ root.end(firing.outcome === "ingress-error" ? "error" : "ok", {
370
+ outcome: firing.outcome,
371
+ ...(firing.runId ? { runId: firing.runId } : {}),
372
+ });
373
+ // Ids, outcome, and the reply's first line only — never a token.
374
+ console.log(
375
+ `[schedule] ${schedule.name} → ${firing.outcome}${firing.runId ? ` run ${firing.runId}` : ""}${firing.detail ? ` — ${firing.detail}` : ""}`,
376
+ );
377
+ // Telemetry for the /runs Scheduled panel — best-effort and off the
378
+ // invocation's critical path: the cron completes when the ingress answered,
379
+ // not when the state Worker has acknowledged the record.
380
+ ctx.waitUntil(recordFiring(env, firing));
381
+ },
382
+ } satisfies ExportedHandler<Env>;