@coreplane/switchboard 0.0.0 → 1.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +18 -1
  3. package/dist/assets/.dockerignore +27 -0
  4. package/dist/assets/.env.example +33 -0
  5. package/dist/assets/Dockerfile +111 -0
  6. package/dist/assets/config/config.example.yaml +359 -0
  7. package/dist/assets/deploy/bin/build-stamp.d.mts +15 -0
  8. package/dist/assets/deploy/bin/build-stamp.mjs +98 -0
  9. package/dist/assets/deploy/bin/cf-logs +32 -0
  10. package/dist/assets/deploy/cloudflare/package.json +29 -0
  11. package/dist/assets/deploy/cloudflare/preflight.mjs +243 -0
  12. package/dist/assets/deploy/cloudflare/tsconfig.json +18 -0
  13. package/dist/assets/deploy/cloudflare/worker.ts +382 -0
  14. package/dist/assets/deploy/cloudflare/wrangler.template.jsonc +67 -0
  15. package/dist/assets/deploy/cloudflare/write-build.d.mts +7 -0
  16. package/dist/assets/deploy/cloudflare/write-build.mjs +53 -0
  17. package/dist/assets/deploy/cloudflare-docs/package.json +18 -0
  18. package/dist/assets/deploy/cloudflare-docs/wrangler.template.jsonc +30 -0
  19. package/dist/assets/deploy/cloudflare-memory/package.json +25 -0
  20. package/dist/assets/deploy/cloudflare-memory/tsconfig.json +17 -0
  21. package/dist/assets/deploy/cloudflare-memory/worker.ts +2635 -0
  22. package/dist/assets/deploy/cloudflare-memory/wrangler.template.jsonc +50 -0
  23. package/dist/assets/deploy/cloudflare-resident/Dockerfile +91 -0
  24. package/dist/assets/deploy/cloudflare-resident/gc.ts +287 -0
  25. package/dist/assets/deploy/cloudflare-resident/node-async-hooks.d.ts +11 -0
  26. package/dist/assets/deploy/cloudflare-resident/package.json +29 -0
  27. package/dist/assets/deploy/cloudflare-resident/preflight.mjs +224 -0
  28. package/dist/assets/deploy/cloudflare-resident/tsconfig.json +19 -0
  29. package/dist/assets/deploy/cloudflare-resident/worker.ts +6637 -0
  30. package/dist/assets/deploy/cloudflare-resident/wrangler.template.jsonc +120 -0
  31. package/dist/assets/deploy/cloudflare-sandbox/Dockerfile +67 -0
  32. package/dist/assets/deploy/cloudflare-sandbox/docker-wrapper.sh +37 -0
  33. package/dist/assets/deploy/cloudflare-sandbox/package.json +26 -0
  34. package/dist/assets/deploy/cloudflare-sandbox/tsconfig.json +20 -0
  35. package/dist/assets/deploy/cloudflare-sandbox/worker.ts +410 -0
  36. package/dist/assets/deploy/cloudflare-sandbox/wrangler.template.jsonc +67 -0
  37. package/dist/assets/deploy/profile.example.json +13 -0
  38. package/dist/assets/deploy/secrets.manifest.json +108 -0
  39. package/dist/assets/docker-entrypoint.sh +15 -0
  40. package/dist/assets/package-lock.json +18407 -0
  41. package/dist/assets/package.json +104 -0
  42. package/dist/assets/project.json +219 -0
  43. package/dist/assets/source.json +5 -0
  44. package/dist/assets/src/core/authz/actor.ts +100 -0
  45. package/dist/assets/src/core/authz/authorize.ts +169 -0
  46. package/dist/assets/src/core/authz/grants.ts +347 -0
  47. package/dist/assets/src/core/authz/policy.ts +281 -0
  48. package/dist/assets/src/core/authz/resource.ts +147 -0
  49. package/dist/assets/src/core/authz/types.ts +164 -0
  50. package/dist/assets/src/core/drain.ts +54 -0
  51. package/dist/assets/src/core/ingressTokens.ts +64 -0
  52. package/dist/assets/src/core/memory/engine.ts +115 -0
  53. package/dist/assets/src/core/memory/scorer.ts +147 -0
  54. package/dist/assets/src/core/memory/types.ts +120 -0
  55. package/dist/assets/src/core/normalizeSpans.ts +299 -0
  56. package/dist/assets/src/core/prDescriptionTypes.ts +54 -0
  57. package/dist/assets/src/core/redact.ts +113 -0
  58. package/dist/assets/src/core/runEvents.ts +537 -0
  59. package/dist/assets/src/core/runFriction.ts +665 -0
  60. package/dist/assets/src/core/runLedger/decisions.ts +126 -0
  61. package/dist/assets/src/core/runLedger/types.ts +177 -0
  62. package/dist/assets/src/core/runRecord.ts +627 -0
  63. package/dist/assets/src/core/runShape.ts +61 -0
  64. package/dist/assets/src/core/schedules.ts +452 -0
  65. package/dist/assets/src/core/time/formatDuration.ts +61 -0
  66. package/dist/assets/src/core/trace/attrs.ts +203 -0
  67. package/dist/assets/src/core/trace/classify.ts +49 -0
  68. package/dist/assets/src/core/trace/clock.ts +6 -0
  69. package/dist/assets/src/core/trace/context.ts +9 -0
  70. package/dist/assets/src/core/trace/ids.ts +23 -0
  71. package/dist/assets/src/core/trace/partition.ts +235 -0
  72. package/dist/assets/src/core/trace/sinks.ts +68 -0
  73. package/dist/assets/src/core/trace/streamSpans.ts +163 -0
  74. package/dist/assets/src/core/trace/traceparent.ts +29 -0
  75. package/dist/assets/src/core/trace/tracer.ts +247 -0
  76. package/dist/assets/src/core/trace/types.ts +125 -0
  77. package/dist/assets/src/core/trace/workerTrace.ts +97 -0
  78. package/dist/assets/src/deploy/buildStamp.ts +93 -0
  79. package/dist/assets/src/deploy/liveGate.ts +203 -0
  80. package/dist/assets/src/deploy/profile.ts +162 -0
  81. package/dist/assets/src/deploy/restart.ts +393 -0
  82. package/dist/assets/src/effort.ts +17 -0
  83. package/dist/assets/src/execution/bashTimeout.ts +78 -0
  84. package/dist/assets/src/execution/bindingPurge.ts +43 -0
  85. package/dist/assets/src/execution/residentBackupTransfer.ts +50 -0
  86. package/dist/assets/src/execution/residentCleanliness.ts +95 -0
  87. package/dist/assets/src/execution/residentCredentials.ts +81 -0
  88. package/dist/assets/src/execution/residentDepCache.ts +321 -0
  89. package/dist/assets/src/execution/residentDepsStore.ts +326 -0
  90. package/dist/assets/src/execution/residentDetach.ts +48 -0
  91. package/dist/assets/src/execution/residentDisk.ts +107 -0
  92. package/dist/assets/src/execution/residentDiskBudget.ts +448 -0
  93. package/dist/assets/src/execution/residentExecWrap.ts +100 -0
  94. package/dist/assets/src/execution/residentHead.ts +85 -0
  95. package/dist/assets/src/execution/residentReadonly.ts +72 -0
  96. package/dist/assets/src/execution/residentRefresh.ts +429 -0
  97. package/dist/assets/src/execution/residentRestoreExtract.ts +130 -0
  98. package/dist/assets/src/execution/residentState.ts +47 -0
  99. package/dist/assets/src/execution/residentStepReport.ts +98 -0
  100. package/dist/assets/src/execution/residentStepTrace.ts +97 -0
  101. package/dist/assets/src/execution/residentSteps.ts +99 -0
  102. package/dist/assets/src/execution/residentText.ts +83 -0
  103. package/dist/assets/src/execution/residentTrace.ts +119 -0
  104. package/dist/assets/src/execution/sandboxEnv.ts +42 -0
  105. package/dist/assets/src/execution/sandboxErrors.ts +159 -0
  106. package/dist/assets/src/execution/sandboxKeepalive.ts +118 -0
  107. package/dist/assets/src/execution/shellQuote.ts +8 -0
  108. package/dist/assets/src/mcp/registry.ts +242 -0
  109. package/dist/assets/src/providers/types.ts +152 -0
  110. package/dist/assets/web/dist/.vite/manifest.json +176 -0
  111. package/dist/assets/web/dist/assets/AppShell-Bk2gbvet.js +1 -0
  112. package/dist/assets/web/dist/assets/CostsPage-CTZcMYYx.js +1 -0
  113. package/dist/assets/web/dist/assets/NotFoundPage-C-BuaSm8.js +1 -0
  114. package/dist/assets/web/dist/assets/ResidentDetailPage-D3shEnzl.js +1 -0
  115. package/dist/assets/web/dist/assets/ResidentsIndexPage-DWIubQ05.js +1 -0
  116. package/dist/assets/web/dist/assets/RunRoutePage-BMjuE-oX.js +126 -0
  117. package/dist/assets/web/dist/assets/RunRoutePage-XVFj0XDc.css +1 -0
  118. package/dist/assets/web/dist/assets/RunsIndexPage-C3_jYIo0.js +1 -0
  119. package/dist/assets/web/dist/assets/RunsTabs-C4krAL9o.js +1 -0
  120. package/dist/assets/web/dist/assets/ScheduledPage-g1W58mtN.js +1 -0
  121. package/dist/assets/web/dist/assets/StatusDot-DcPRw3zu.js +1 -0
  122. package/dist/assets/web/dist/assets/Tooltip-DJUkMYjo.js +1 -0
  123. package/dist/assets/web/dist/assets/favicon-DL1rdWJt.js +1 -0
  124. package/dist/assets/web/dist/assets/localIso-L06jV29p.js +1 -0
  125. package/dist/assets/web/dist/assets/main-BsBGUyMH.css +2 -0
  126. package/dist/assets/web/dist/assets/main-CyM5f4JC.js +28 -0
  127. package/dist/assets/web/dist/assets/residentDiskBudget-BMBKlYRH.js +1 -0
  128. package/dist/assets/web/dist/assets/seed-BglCRKLA.js +6 -0
  129. package/dist/assets/web/dist/assets/wallClock-Ckv3sKoR.js +1 -0
  130. package/dist/cli.js +34494 -0
  131. package/package.json +43 -10
@@ -0,0 +1,72 @@
1
+ /** The read-only attach decision of the resident Worker's `attachThread`
2
+ * (deploy/cloudflare-resident/worker.ts), kept pure and dependency-free so it
3
+ * is unit-testable from src/ and imported across packages by the resident
4
+ * Worker (like residentDetach / shellQuote) — the tested code IS the shipped
5
+ * code.
6
+ *
7
+ * Background: a review run — read-only "by convention" — held a per-attach
8
+ * GitHub credential file and an `origin` pointing at GitHub, fetched another
9
+ * PR's branch because the PR body referenced it, and its verdict landed on
10
+ * the wrong PR (the reviewed-head guard on the post step is the other half
11
+ * of the fix). This removes the capability: a read-only attach gets no
12
+ * credential file and an
13
+ * `origin` it cannot fetch from, so "read-only" is enforced by the worktree,
14
+ * not requested of the model.
15
+ *
16
+ * Mode is sticky per attach, not per thread: the binding records the mode
17
+ * the tree was last built for, and an attach in the OTHER mode recreates the
18
+ * tree (a credential-less, mirror-origin tree must never be handed to a
19
+ * writable run, and a writable tree's credential file must never survive
20
+ * into a read-only run). Recreating is the simple safe option — the only
21
+ * state a tree carries between attaches is scratch files, and a mode switch
22
+ * on one thread is rare (a thread is normally one agent for its whole life). */
23
+
24
+ export type ParsedReadonly = { readonly: boolean } | { error: string };
25
+
26
+ /** `/attach` body field `readonly`: absent → false (older bots never send it);
27
+ * a boolean → itself; anything else → a 400-shaped error. */
28
+ export function parseReadonly(value: unknown): ParsedReadonly {
29
+ if (value === undefined) return { readonly: false };
30
+ if (typeof value === "boolean") return { readonly: value };
31
+ return { error: "readonly must be a boolean when present" };
32
+ }
33
+
34
+ export interface ReadonlyAttachPlan {
35
+ /** The mode this attach builds/reuses the tree for (recorded on the binding). */
36
+ readonly: boolean;
37
+ /** True when the existing tree was built for the other mode and must be wiped
38
+ * before reuse — evaluated before the ordinary dirty/stale checks. */
39
+ modeSwitch: boolean;
40
+ /** Mint a repo-scoped token for the tree and write the credential file (writable only). A read-only attach never puts a token in the tree; the mirror's own recovery fetch (root, outside the tree) may still mint one. */
41
+ credentialFile: boolean;
42
+ /** Remove any credential file / helper config from the tree (read-only, every
43
+ * attach — a reused tree may predate this rule). */
44
+ scrubCredentials: boolean;
45
+ /** What the worktree's `origin` points at after clone. */
46
+ originUrl: string;
47
+ }
48
+
49
+ export function planReadonlyAttach(input: {
50
+ readonly: boolean;
51
+ /** The thread's existing binding, if any. `evicted` trees are gone from disk
52
+ * and are recreated anyway, so their recorded mode is irrelevant. */
53
+ prior?: { readonly?: boolean; evicted?: boolean };
54
+ /** `owner/name` of the repo — the writable origin. */
55
+ slug: string;
56
+ /** The resident's bare mirror path — the read-only origin. Thread users are
57
+ * denied traversal into it (root:worker1 750), so fetch/push fail while
58
+ * the clone-time remote-tracking refs (`origin/<default>`) stay usable for
59
+ * `git diff origin/main...HEAD`. */
60
+ mirrorDir: string;
61
+ }): ReadonlyAttachPlan {
62
+ const { readonly, prior, slug, mirrorDir } = input;
63
+ const priorLive = prior !== undefined && !prior.evicted;
64
+ const modeSwitch = priorLive && (prior.readonly ?? false) !== readonly;
65
+ return {
66
+ readonly,
67
+ modeSwitch,
68
+ credentialFile: !readonly,
69
+ scrubCredentials: readonly,
70
+ originUrl: readonly ? mirrorDir : `https://github.com/${slug}.git`,
71
+ };
72
+ }
@@ -0,0 +1,429 @@
1
+ /** The refresh cycle's rebuild decision for the resident Worker's
2
+ * `onRefreshAlarm` (deploy/cloudflare-resident/worker.ts), kept pure and
3
+ * dependency-free so it is unit-testable from src/ and imported across
4
+ * packages by the resident Worker (like residentDetach) — the tested code IS
5
+ * the shipped code.
6
+ *
7
+ * Background: the refresh used to `git clean -fdx` + `npm install` on
8
+ * EVERY default-branch advance (a refresh over two minutes long), even
9
+ * though the committed lockfile — the dependency cache key — rarely
10
+ * moves. A refresh longer than the bot's 60 s attach wait sends runs cold.
11
+ * The `-x` clean itself is load-bearing: attached, sha-pinned thread
12
+ * worktrees hardlink the checkout's dep/build FILE inodes, so a rebuild must
13
+ * allocate fresh inodes rather than write through shared ones (review 1b).
14
+ * Skipping install therefore keeps `node_modules` — inodes nobody is about
15
+ * to write through — while still cleaning every other gitignored path (build
16
+ * output), so the build allocates fresh inodes exactly as before.
17
+ *
18
+ * The disk facts come from two root-only markers the cycle writes as it
19
+ * materializes (deps key after install, built sha after build) plus the
20
+ * checkout's actual HEAD. A worker-code deploy resets the DO mid-cycle but
21
+ * leaves the container disk alone, so the next alarm can pick up where the
22
+ * interrupted one stopped instead of redoing the whole rebuild. */
23
+
24
+ import { DISK_FULL_FREE_KIB, diskFullReason, isDiskFullMessage } from "./residentDisk.js";
25
+
26
+ export interface RefreshDisk {
27
+ /** `git rev-parse HEAD` of the warm checkout; null when unreadable. */
28
+ head: string | null;
29
+ /** Lockfile key whose install fully completed into the checkout's node_modules; null when absent. */
30
+ installedKey: string | null;
31
+ /** Lockfile key an install was STARTED for and never completed (the marker is
32
+ * written before `install` and removed after the deps key lands); null when absent. */
33
+ installingKey: string | null;
34
+ /** Sha whose build fully completed in the checkout; null when absent. */
35
+ builtSha: string | null;
36
+ }
37
+
38
+ /** What the `-x` clean may remove: everything untracked, or everything except `node_modules`. */
39
+ export type CleanScope = "all" | "keep-deps";
40
+
41
+ export type RefreshPlan =
42
+ /** The default branch did not move — nothing to rebuild or snapshot. */
43
+ | { action: "unchanged" }
44
+ /** Checkout, deps and build already match the target (an interrupted cycle
45
+ * got this far): skip straight to the snapshot. */
46
+ | { action: "reuse"; why: string }
47
+ /** Update the checkout; `install` only when the committed lockfile moved. */
48
+ | { action: "rebuild"; install: boolean; clean: CleanScope; why: string };
49
+
50
+ export function planRefresh(input: {
51
+ /** Mirror sha of the default branch after the fetch. */
52
+ sha: string;
53
+ /** The sha the recorded facts (and the last snapshot) are at. */
54
+ factsSha: string;
55
+ /** Committed-lockfile key at `sha` (pure function of the commit). */
56
+ lockfileKey: string;
57
+ disk: RefreshDisk;
58
+ }): RefreshPlan {
59
+ const { sha, factsSha, lockfileKey, disk } = input;
60
+ if (sha === factsSha) return { action: "unchanged" };
61
+ const depsMatch = disk.installedKey !== null && disk.installedKey === lockfileKey;
62
+ if (depsMatch && disk.head === sha && disk.builtSha === sha) {
63
+ return { action: "reuse", why: `checkout, deps and build already materialized for ${sha.slice(0, 8)}` };
64
+ }
65
+ if (depsMatch) {
66
+ return { action: "rebuild", install: false, clean: "keep-deps", why: "lockfile unchanged — deps kept, build only" };
67
+ }
68
+ if (disk.installedKey === null && disk.installingKey !== null && disk.installingKey === lockfileKey) {
69
+ // A previous cycle started this very install and ended before the deps
70
+ // key landed (its step timed out, or its DO isolate died; the pre-step
71
+ // sweep has killed any writer it left). npm reconciles a partial tree to
72
+ // the lockfile, so resuming converges where wipe-and-restart cannot: a
73
+ // repo whose cold install outruns one step budget still lands over cycles
74
+ // (wipe-and-restart runs full installs back to back, none finishing).
75
+ // Safe for the hardlink invariant (review 1b): threads only link deps
76
+ // whose key the deps marker vouches for, and no marker vouched for these.
77
+ return {
78
+ action: "rebuild",
79
+ install: true,
80
+ clean: "keep-deps",
81
+ why: "install resumes — a previous attempt for this lockfile ended before the deps marker; npm reconciles the partial tree",
82
+ };
83
+ }
84
+ const why = disk.installedKey === null ? "no deps marker on disk — full install" : "lockfile changed — full install";
85
+ return { action: "rebuild", install: true, clean: "all", why };
86
+ }
87
+
88
+ /** Tool caches that BUILDS (not installs) write inside node_modules — and
89
+ * typically open+truncate in place: babel-loader/eslint/webpack under
90
+ * `.cache`, vite/vitest under `.vite`. Kept deps are hardlinked into attached
91
+ * worktrees, so these must go before a keep-deps build or it would write
92
+ * through shared inodes (review 1b). Pruned at any depth (workspaces). */
93
+ export const NODE_MODULES_CACHE_DIRS = [".cache", ".vite"] as const;
94
+
95
+ /** The checkout-update shell for the build user (runs inside the checkout):
96
+ * fetch from the local mirror, hard-reset to `sha`, then the `-x` clean.
97
+ * `-e node_modules` is a git exclude pattern (matches at any depth, so
98
+ * workspace packages keep theirs too) that survives `-x`; everything else
99
+ * gitignored — build output above all — is still removed so the build
100
+ * allocates fresh inodes (review 1b). Keep-deps additionally sweeps the
101
+ * build-written caches inside node_modules (NODE_MODULES_CACHE_DIRS). */
102
+ export function checkoutUpdateCommand(sha: string, clean: CleanScope): string {
103
+ const base = `git fetch --quiet origin && git reset --hard --quiet ${sha}`;
104
+ if (clean === "all") return `${base} && git clean -fdx`;
105
+ const names = NODE_MODULES_CACHE_DIRS.map((d) => `-name ${d}`).join(" -o ");
106
+ const sweep = `find . -path '*/node_modules/*' -type d \\( ${names} \\) -prune -exec rm -rf {} +`;
107
+ return `${base} && git clean -fdx -e node_modules && ${sweep}`;
108
+ }
109
+
110
+ // -- interruption vs. failure ------------------------------------------------
111
+
112
+ /** How a refresh-cycle step failed. `interrupted` is the one outcome that says
113
+ * NOTHING about the repository: the step was killed from outside because the
114
+ * CONTAINER was replaced under it — an image-changing deploy or an explicit
115
+ * container stop/restart. (A Worker-only deploy swaps the DO isolate but
116
+ * leaves the container and its processes running, so it cannot interrupt a
117
+ * step at all.) The kill surfaces either as the shell's
118
+ * own death (SIGTERM, exit 143, "Session terminated") or, past the shell, as
119
+ * the SDK's replacement errors (stale process handle, closed supervisor).
120
+ * Everything else is the repo's own build failing. */
121
+ export interface RefreshFailure {
122
+ /** The `degraded` reason to record. Interruptions are prefixed
123
+ * `refresh-interrupted:` so the park-streak gate can exclude them by prefix,
124
+ * exactly like the watchdog's stamps; a full disk is `disk-full:`
125
+ * (`residentDisk.ts` — the cycle's entry gate and the recycle decision key
126
+ * on it); real failures keep `<step>-failed:`. */
127
+ reason: string;
128
+ interrupted: boolean;
129
+ diskFull: boolean;
130
+ }
131
+
132
+ /** Root argv that kills every process the build user still owns and waits
133
+ * (bounded, 5 s) until none is left. Runs before EVERY build-user step.
134
+ *
135
+ * Why: an `npm install` that outlives its budget and the SDK's output grace
136
+ * is recorded as a timeout while npm keeps extracting into
137
+ * CHECKOUT_DIR/node_modules. The next cycle's `git clean -fdx` races it —
138
+ * `warning: failed to remove node_modules/<pkg>: Directory not empty` on
139
+ * exactly the packages being written — and the resident spirals between
140
+ * `checkout-update-failed` and install timeouts (each cycle's install now
141
+ * sharing 1 vCPU with the last one's orphan) for as long as the default
142
+ * branch keeps moving. A Worker-only deploy is the other way to orphan a
143
+ * step: the DO isolate resets, the container keeps running.
144
+ *
145
+ * Scoped to the TREE the step is about to touch (`dir`): a process counts as
146
+ * stale when its cwd is `dir` or below it. Steps on one tree are strictly
147
+ * sequential, so a live build-user process inside that tree at step start
148
+ * is by definition a leftover — while the same user's installs into OTHER
149
+ * trees (the deps store runs distinct keys in parallel, item 59) are live
150
+ * work this sweep must not touch. Survivors are NAMED on stdout (the Worker
151
+ * log shows what was still running) before SIGKILL: SIGTERM would let npm
152
+ * keep writing while the clean runs. Nothing matching is the happy path; a
153
+ * process that survives SIGKILL for 5 s fails the step — nothing may start
154
+ * beside it. */
155
+ export function killStaleBuildProcessesCommand(user: string, dir: string): string[] {
156
+ const inTree = `case "$(readlink /proc/$p/cwd 2>/dev/null)" in ${dir}|${dir}/*) `;
157
+ const script =
158
+ `stale=""; for p in $(pgrep -u ${user}); do ${inTree}stale="$stale$p ";; esac; done; ` +
159
+ `if [ -n "$stale" ]; then ` +
160
+ `echo "killing stale ${user} processes under ${dir}:"; ps -o pid=,args= -p $(echo $stale | tr ' ' ',') 2>/dev/null; ` +
161
+ `kill -KILL $stale 2>/dev/null; ` +
162
+ `i=0; while :; do alive=""; for p in $stale; do kill -0 $p 2>/dev/null && alive="$alive$p "; done; ` +
163
+ `[ -z "$alive" ] && break; i=$((i+1)); ` +
164
+ `if [ $i -ge 50 ]; then echo "stale ${user} processes survived SIGKILL for 5s: $alive" >&2; exit 1; fi; ` +
165
+ `sleep 0.1; done; ` +
166
+ `fi`;
167
+ return ["sh", "-c", script];
168
+ }
169
+
170
+ /** Signature of a step killed from OUTSIDE its own budget: the exit status of
171
+ * SIGTERM (128 + 15), bash's "Session terminated" on a killed login shell, or
172
+ * a tool naming the signal. A bare "killed" is NOT enough — compilers and
173
+ * OOM messages say it too — and a step the cycle itself timed out is a real
174
+ * failure however it died. */
175
+ const INTERRUPTION_SIGNATURE = /\bexit 143\b|Session terminated|SIGTERM/;
176
+
177
+ /** Message wording of the Sandbox SDK's runtime-replacement error family — the
178
+ * container went away UNDER a live SDK call, so the failure never reaches the
179
+ * shell-kill signature above: a step that dies this way surfaces as
180
+ * `StaleProcessHandleError` ("previous runtime incarnation"),
181
+ * `ProcessSpawnFailedError` ("Process supervisor is closed"), one of the SDK's
182
+ * interruption messages, or — when a deploy ROLLS the container out from under
183
+ * the run rather than just swapping the DO isolate — the raw workerd binding
184
+ * refusal "The container is not running, consider calling start()". That
185
+ * last one is a spawn-phase refusal: workerd rejected the process start because
186
+ * the container was not running at all, so nothing launched (safe to re-attach
187
+ * and let the model re-check). It is NOT a typed SDK error (the SDK's own
188
+ * auto-start path re-throws it raw when a roll outlasts its port-ready bound)
189
+ * and NOT the `container_stopped` `OperationInterruptedError` reason (that is a
190
+ * stop UNDER an in-flight op, already covered by the typed check), so the
191
+ * message wording is the only signal — deliberately anchored to the full
192
+ * "consider calling start" phrase so it can never match a genuine container
193
+ * crash ("container exited with unexpected exit code") or the readiness probe
194
+ * ("the container is not listening"), which must stay ordinary failures.
195
+ * Shared with the resident Worker's `isRuntimeReplacement`
196
+ * (deploy/cloudflare-resident/worker.ts) as its message-level fallback, so the
197
+ * exec path and the refresh classifier agree on one wording list (a container
198
+ * stop mid-snapshot produces "Process supervisor is closed"; classified as the
199
+ * repo's own `snapshot-failed` it would re-arm at the full cadence). */
200
+ export const RUNTIME_REPLACEMENT_WORDING =
201
+ /previous runtime incarnation|interrupted because the runtime changed|runtime identity is no longer active|sandbox lifetime is no longer current|platform was updating the sandbox runtime|no longer identifies pid|process supervisor is closed|container is not running, consider calling start/i;
202
+
203
+ /** `freeKiB` is the `df` probe's answer (`parseDfFreeKiB`), taken AFTER the
204
+ * step failed and only consulted when the message itself carries no errno:
205
+ * a message saying ENOSPC is disk-full outright (the disk is the actionable
206
+ * fact, whatever else the message says); a kill signature without it is an
207
+ * interruption whatever the disk holds (the kill ended the step); otherwise a
208
+ * probe below the floor names the disk, and no probe (`undefined`/`null`)
209
+ * leaves the step's own failure — unknown is never full. */
210
+ export function classifyRefreshFailure(input: {
211
+ step: string;
212
+ message: string;
213
+ freeKiB?: number | null;
214
+ }): RefreshFailure {
215
+ const { step, message } = input;
216
+ if (isDiskFullMessage(message)) {
217
+ return {
218
+ interrupted: false,
219
+ diskFull: true,
220
+ reason: diskFullReason({ step, message, freeKiB: input.freeKiB ?? null }),
221
+ };
222
+ }
223
+ const timedOut = /\(timed out\)/.test(message);
224
+ if (!timedOut && (INTERRUPTION_SIGNATURE.test(message) || RUNTIME_REPLACEMENT_WORDING.test(message))) {
225
+ return { interrupted: true, diskFull: false, reason: `refresh-interrupted: ${step} ${message}` };
226
+ }
227
+ if (input.freeKiB !== undefined && input.freeKiB !== null && input.freeKiB < DISK_FULL_FREE_KIB) {
228
+ return { interrupted: false, diskFull: true, reason: diskFullReason({ step, message, freeKiB: input.freeKiB }) };
229
+ }
230
+ return { interrupted: false, diskFull: false, reason: `${step}-failed: ${message}` };
231
+ }
232
+
233
+ /** Re-arm delay after a cycle that did not run to completion for a reason
234
+ * outside the repo — an interrupted step, or an image-stale container stop.
235
+ * 45 s: comfortably longer than a container restart plus rehydration
236
+ * (~10–20 s observed), so the retry finds a live runtime, and an order of
237
+ * magnitude under the 600 s cadence that previously left the resident
238
+ * `degraded` (every run falling back cold) until the next regular alarm. */
239
+ export const INTERRUPTED_REARM_S = 45;
240
+
241
+ export type RefreshOutcome =
242
+ /** Cycle ran (warm, or a real failure): regular cadence. */
243
+ | "normal"
244
+ /** A step was killed from outside (see classifyRefreshFailure). */
245
+ | "interrupted"
246
+ /** `reconcileImage` stopped the container so it restarts on the new image. */
247
+ | "image-stale-restart"
248
+ /** The disk-full recovery stopped the container so it restarts on an empty
249
+ * disk and the next alarm restores from R2 (`residentDisk.ts`). */
250
+ | "disk-full-restart"
251
+ /** Idle gate parked the resident. */
252
+ | "idle";
253
+
254
+ /** How many CONSECUTIVE interrupted cycles still re-arm short. A real deploy
255
+ * interrupts once, maybe twice (a deploy train); a step whose own output
256
+ * happens to carry the kill signature every cycle (a test supervisor printing
257
+ * `signal SIGTERM`) would otherwise retry at 45 s forever — never parking
258
+ * (it is excluded from the streak) AND at 13× the cadence. Past the cap the
259
+ * regular interval returns; the classification (and the streak exclusion)
260
+ * stand, matching the watchdog's accepted "never parks, but at cadence". */
261
+ export const INTERRUPTED_REARM_MAX_CONSECUTIVE = 3;
262
+
263
+ export function nextRefreshDelayS(input: {
264
+ outcome: RefreshOutcome;
265
+ intervalS: number;
266
+ idleIntervalS: number;
267
+ /** Consecutive cycles (this one included) that ended `interrupted`; omitted = 1. */
268
+ consecutiveInterrupted?: number;
269
+ }): number {
270
+ switch (input.outcome) {
271
+ case "idle":
272
+ return input.idleIntervalS;
273
+ case "interrupted":
274
+ return (input.consecutiveInterrupted ?? 1) > INTERRUPTED_REARM_MAX_CONSECUTIVE
275
+ ? input.intervalS
276
+ : INTERRUPTED_REARM_S;
277
+ case "image-stale-restart":
278
+ case "disk-full-restart":
279
+ return INTERRUPTED_REARM_S;
280
+ default:
281
+ return input.intervalS;
282
+ }
283
+ }
284
+
285
+ /** Bound a promise that offers no timeout of its own (the
286
+ * Sandbox SDK's createBackup/restoreBackup take neither a timeout nor an
287
+ * AbortSignal). On expiry, rejects with an error naming `what` and the
288
+ * budget, so a hung R2 transfer fails the refresh cycle into its existing
289
+ * degrade/goDown handling instead of stranding `refreshing`/`restoring`
290
+ * until the 30-min watchdog. The losing promise keeps running (nothing can
291
+ * cancel it) — its eventual rejection is swallowed so it never surfaces as
292
+ * an unhandled rejection. */
293
+ // -- restore progress ---------------------------------------------------------------
294
+
295
+ /** How often the wake path samples a restore's target directory. */
296
+ export const RESTORE_POLL_MS = 15_000;
297
+ /** A restore whose target has not grown for this long is stalled. Generous
298
+ * against R2's own hiccups, tight against a hung SDK operation: a healthy
299
+ * restore writes continuously, even when the same ~2 GiB checkout snapshot
300
+ * takes anywhere from under a minute to eight minutes. */
301
+ export const RESTORE_STALL_MS = 120_000;
302
+ /** Absolute cap for ONE HYDRATE — the wait for a previous attempt's restore,
303
+ * the mirror restore and the checkout restore share it (one deadline, see
304
+ * `deadlineMs`) — under the watchdog's 30-min stale-mid-flight window so a
305
+ * runaway wake is still the wake path's own verdict, not the watchdog's. */
306
+ export const RESTORE_MAX_MS = 25 * 60_000;
307
+
308
+ /** Below this much of the hydrate deadline left, the wake does not start a
309
+ * deps materialization at all: a download or install that cannot finish is
310
+ * worse than the next refresh cycle's repair (`no deps marker on disk`). */
311
+ export const WAKE_DEPS_MIN_MS = 60_000;
312
+
313
+ export type WakeDepsBudget =
314
+ | { action: "skip"; remainingMs: number }
315
+ | { action: "materialize"; installBudgetMs: number; restoreDeadlineMs: number; remainingMs: number };
316
+
317
+ /** The wake path's deps materialization (item 61 PR B: restore the warm
318
+ * key's entry, or install it) lives INSIDE the hydrate's one deadline, so
319
+ * the worst-case `restoring` span is still RESTORE_MAX_MS — the invariant
320
+ * the watchdog's 30-min stale-mid-flight window rests on. The restore is
321
+ * judged against the hydrate deadline itself; the installer's budget is the
322
+ * smaller of its own and what remains; under WAKE_DEPS_MIN_MS nothing starts. */
323
+ export function planWakeDepsBudget(input: {
324
+ nowMs: number;
325
+ deadlineMs: number;
326
+ installBudgetMs: number;
327
+ }): WakeDepsBudget {
328
+ const remainingMs = Math.max(0, input.deadlineMs - input.nowMs);
329
+ if (remainingMs < WAKE_DEPS_MIN_MS) return { action: "skip", remainingMs };
330
+ return {
331
+ action: "materialize",
332
+ installBudgetMs: Math.min(input.installBudgetMs, remainingMs),
333
+ restoreDeadlineMs: input.deadlineMs,
334
+ remainingMs,
335
+ };
336
+ }
337
+
338
+ /** Where the Sandbox SDK stages a backup archive inside the container while it
339
+ * downloads (`BACKUP_CONTAINER_DIR` in @cloudflare/sandbox): the restore
340
+ * writes `<dir>/<backupId>.sqsh` in full FIRST and extracts into the target
341
+ * only afterwards, so a restore's progress lives here during the download and
342
+ * in the target during the extraction. Pinned by a test that reads the
343
+ * installed SDK's constant, so an SDK bump that moves it fails the build. */
344
+ export const SDK_BACKUP_ARCHIVE_DIR = "/var/backups";
345
+
346
+ export function restoreArchivePath(backupId: string): string {
347
+ return `${SDK_BACKUP_ARCHIVE_DIR}/${backupId}.sqsh`;
348
+ }
349
+
350
+ export interface RestoreSample {
351
+ atMs: number;
352
+ /** `du -xsk <dir>` at that moment; null when du could not answer. */
353
+ kiB: number | null;
354
+ }
355
+
356
+ export type RestoreVerdict = { verdict: "wait" } | { verdict: "stalled" | "capped"; detail: string };
357
+
358
+ /** Judge a running restore by its bytes, not by a clock. The Sandbox SDK's
359
+ * restoreBackup accepts no timeout, progress callback or AbortSignal, and a
360
+ * promise abandoned by a fixed budget keeps writing (a checkout restore that
361
+ * outlives a fixed 300 s budget has already sent the resident
362
+ * `down(r2-restore-failed)`, and the next hydrate runs `rm -rf` over the tree
363
+ * the first one is still filling). So the wake path polls the
364
+ * target directory: while bytes keep arriving it waits — a slow transfer is
365
+ * a slow transfer — and it gives up only when nothing has been written for
366
+ * RESTORE_STALL_MS (the clock runs from the start until the first byte) or
367
+ * the whole thing exceeds RESTORE_MAX_MS. A sample du could not take is no
368
+ * evidence either way: it neither counts as growth nor resets the clock. */
369
+ export function judgeRestoreProgress(input: {
370
+ startedMs: number;
371
+ nowMs: number;
372
+ samples: readonly RestoreSample[];
373
+ stallMs?: number;
374
+ /** Absolute cap shared by the WHOLE hydrate — the wait for a previous
375
+ * attempt's restore, the mirror restore and the checkout restore all judge
376
+ * against the same instant, so the sum stays under the watchdog's window.
377
+ * Defaults to this restore's start + RESTORE_MAX_MS. */
378
+ deadlineMs?: number;
379
+ }): RestoreVerdict {
380
+ const stallMs = input.stallMs ?? RESTORE_STALL_MS;
381
+ const deadlineMs = input.deadlineMs ?? input.startedMs + RESTORE_MAX_MS;
382
+ const elapsedMs = input.nowMs - input.startedMs;
383
+ let highKiB = 0;
384
+ let lastGrowthMs = input.startedMs;
385
+ for (const s of input.samples) {
386
+ if (s.kiB !== null && s.kiB > highKiB) {
387
+ highKiB = s.kiB;
388
+ lastGrowthMs = s.atMs;
389
+ }
390
+ }
391
+ const gib = (kiB: number) => `${(kiB / 1_048_576).toFixed(2)} GiB`;
392
+ const secs = (ms: number) => `${Math.round(ms / 1000)} s`;
393
+ if (input.nowMs > deadlineMs) {
394
+ return {
395
+ verdict: "capped",
396
+ detail: `still restoring after ${secs(elapsedMs)} (${gib(highKiB)} written) — the hydrate's ${secs(RESTORE_MAX_MS)} cap passed`,
397
+ };
398
+ }
399
+ // Idle runs against NOW from the last observed growth (or the start): a
400
+ // sample du could not take proves nothing, so it neither resets nor pauses
401
+ // the clock — a stall window without evidence of progress is a stall.
402
+ const idleMs = input.nowMs - lastGrowthMs;
403
+ if (idleMs > stallMs) {
404
+ return {
405
+ verdict: "stalled",
406
+ detail: `no bytes written for ${secs(idleMs)} (${gib(highKiB)} after ${secs(elapsedMs)})`,
407
+ };
408
+ }
409
+ return { verdict: "wait" };
410
+ }
411
+
412
+ export function withTimeout<T>(promise: Promise<T>, ms: number, what: string): Promise<T> {
413
+ return new Promise<T>((resolve, reject) => {
414
+ const timer = setTimeout(() => {
415
+ promise.catch(() => {});
416
+ reject(new Error(`${what} timed out after ${ms}ms`));
417
+ }, ms);
418
+ promise.then(
419
+ (v) => {
420
+ clearTimeout(timer);
421
+ resolve(v);
422
+ },
423
+ (e: unknown) => {
424
+ clearTimeout(timer);
425
+ reject(e as Error);
426
+ },
427
+ );
428
+ });
429
+ }
@@ -0,0 +1,130 @@
1
+ /** Extracting an R2 restore onto the resident's own disk (docs/reference/specs/resident-repos.md
2
+ * item 61), kept pure and dependency-free so it is unit-testable from
3
+ * src/ and imported by the resident Worker — the tested code IS the shipped
4
+ * code.
5
+ *
6
+ * Background: in presigned mode the Sandbox SDK's restore does
7
+ * not extract the archive, it MOUNTS it — squashfuse on the `.sqsh` under
8
+ * `/var/backups/mounts/<id>_<ts>_<rand>/lower`, then fuse-overlayfs at the
9
+ * handle's `dir` with a writable upper next to it — and leaves the `.sqsh` in
10
+ * `/var/backups`. (Extraction with `unsquashfs` is the SDK's LOCAL-DEV path.)
11
+ * A mount point breaks code that expects a directory on one ext4 filesystem:
12
+ * `rm -rf /workspace/mirror` → "Device or resource busy" (a resident loops
13
+ * `degraded` on its wake path), `chown -R` → a copy-up of every inode (past
14
+ * any step budget), `du -x` → the
15
+ * mirror and checkout measured as ~1 MiB, and every hardlink or rename
16
+ * between the checkout and the deps store crossed devices.
17
+ *
18
+ * So the resident restores INTO A STAGING MOUNT (a sibling of the target,
19
+ * still under /workspace where the SDK allows it), extracts from it onto the
20
+ * real disk — `unsquashfs` straight from the downloaded `.sqsh` when the image
21
+ * has squashfs-tools (the Dockerfile installs it), `cp -a` out of the mount
22
+ * while an older image lacks it — unmounts the staging mount and its lower,
23
+ * removes the archive, and only then renames the extracted tree into place.
24
+ * After a wake the disk looks exactly as it did before presigned mode. Every
25
+ * clean step first unmounts whatever a previous incarnation left behind. */
26
+
27
+ import { shellQuote } from "./shellQuote.js";
28
+
29
+ /** Where the SDK keeps a mounted restore's lower/upper/work dirs. */
30
+ export const RESTORE_MOUNT_ROOT = "/var/backups/mounts";
31
+
32
+ /** The staging mount for restoring `targetDir`: a sibling, unique per attempt.
33
+ * Must stay under /workspace (or the SDK's other allowed roots) because the
34
+ * SDK validates the handle's `dir`. */
35
+ export function restoreMountDir(targetDir: string, attempt: string): string {
36
+ return `${targetDir}.restore-${attempt}`;
37
+ }
38
+
39
+ /** A backup id as the SDK mints it (a UUID: hex and hyphens, at least one hex
40
+ * digit) — it becomes a glob under the mount root, so nothing else may. */
41
+ const BACKUP_ID_RE = /^(?=.*[0-9a-fA-F])[0-9a-fA-F-]{8,64}$/;
42
+
43
+ /** Unmount ONE staging mount and what hangs off it, as root, tolerating a
44
+ * path that is not (or no longer) mounted: the fuse-overlayfs at `mountDir`
45
+ * first, then every squashfuse lower under the SDK's `<backupId>_<ts>_<rand>`
46
+ * dirs, then those dirs. The lower is found by the backup id, not by the
47
+ * overlay's options: fuse-overlayfs exposes no `lowerdir=` in /proc/mounts
48
+ * (an unmount keyed on the overlay's options leaves the squashfuse lowers
49
+ * mounted, each pinning its unlinked `.sqsh`).
50
+ * `fusermount3 -u` is the FUSE way; `umount -l` the fallback for a busy
51
+ * mount. */
52
+ export function unmountRestoreScript(input: { mountDir: string; backupId: string }): string {
53
+ if (!BACKUP_ID_RE.test(input.backupId))
54
+ throw new Error(`restore: not a backup id: ${JSON.stringify(input.backupId)}`);
55
+ const m = shellQuote(input.mountDir);
56
+ const lowers = `${RESTORE_MOUNT_ROOT}/${input.backupId}_*/lower`;
57
+ return [
58
+ `_m=${m}`,
59
+ `if awk -v m="$_m" '$2==m {f=1} END {exit !f}' /proc/mounts 2>/dev/null; then fusermount3 -u "$_m" 2>/dev/null || umount -l "$_m" 2>/dev/null || true; fi`,
60
+ `for _l in ${lowers}; do`,
61
+ ` if awk -v m="$_l" '$2==m {f=1} END {exit !f}' /proc/mounts 2>/dev/null; then fusermount3 -u "$_l" 2>/dev/null || umount -l "$_l" 2>/dev/null || true; fi`,
62
+ `done`,
63
+ // The removals are best effort: after a LAZY unmount a still-open file
64
+ // can keep the dir alive for a moment, and this script also runs under
65
+ // `set -e` inside extractRestoreScript — a successful extraction must
66
+ // never be failed by its cleanup. The unmount-all clean step reclaims
67
+ // whatever is left. The glob takes EVERY mount dir of this backup id, not
68
+ // only this attempt's: the SDK serializes backup operations on one queue,
69
+ // so no other restore of the same archive can be mid-flight here, and an
70
+ // earlier attempt's dir for the id is exactly the debris to remove.
71
+ `rm -rf ${RESTORE_MOUNT_ROOT}/${input.backupId}_* 2>/dev/null || true`,
72
+ `rmdir "$_m" 2>/dev/null || rm -rf "$_m" 2>/dev/null || true`,
73
+ ].join("\n");
74
+ }
75
+
76
+ /** Turn a mounted restore into a plain directory tree, as root:
77
+ * 1. extract — `unsquashfs` from the `.sqsh` the SDK downloaded when the
78
+ * image has it (multi-threaded, no FUSE in the read path; `-no-xattrs`
79
+ * because squashfs xattrs are not worth a failed restore), else `cp -a`
80
+ * out of the mount (ownership and modes preserved either way — the
81
+ * archive was taken from a worker1-owned tree, so no chown follows);
82
+ * 2. unmount the staging mount and its lower, remove the archive;
83
+ * 3. rename the extracted tree to the target — the target appears LAST, so
84
+ * a failure anywhere above leaves no half-populated target.
85
+ * Prints `extract: unsquashfs` or `extract: cp` so the log says which. */
86
+ export function extractRestoreScript(input: {
87
+ mountDir: string;
88
+ backupId: string;
89
+ archivePath: string;
90
+ targetDir: string;
91
+ }): string {
92
+ const attempt = input.mountDir.slice(input.mountDir.lastIndexOf(".restore-") + ".restore-".length);
93
+ const tmp = shellQuote(`${input.targetDir}.extract-${attempt}`);
94
+ const archive = shellQuote(input.archivePath);
95
+ const target = shellQuote(input.targetDir);
96
+ return [
97
+ `set -e`,
98
+ `rm -rf ${tmp}`,
99
+ `if command -v unsquashfs >/dev/null 2>&1 && test -f ${archive}; then`,
100
+ ` unsquashfs -n -no-xattrs -d ${tmp} ${archive} >/dev/null`,
101
+ ` echo "extract: unsquashfs"`,
102
+ `else`,
103
+ ` mkdir ${tmp}`,
104
+ ` cp -a ${shellQuote(`${input.mountDir}/.`)} ${tmp}/`,
105
+ ` echo "extract: cp"`,
106
+ `fi`,
107
+ unmountRestoreScript({ mountDir: input.mountDir, backupId: input.backupId }),
108
+ `rm -f ${archive}`,
109
+ `rm -rf ${target}`,
110
+ `mv ${tmp} ${target}`,
111
+ ].join("\n");
112
+ }
113
+
114
+ /** For the clean steps (`clean-before-restore`, provisioning's
115
+ * `clean-workspace`): unmount every fuse mount a previous incarnation left
116
+ * under /workspace or under the SDK's mount root — deepest first, so an
117
+ * overlay goes before the lower it sits on — then remove the SDK's mount dirs,
118
+ * any `.sqsh` still on disk, and the staging (`*.restore-*`) and extraction
119
+ * (`*.extract-*`) siblings an attempt that died mid-way left beside the
120
+ * mirror or checkout (a multi-GiB partial tree on a disk-budgeted resident).
121
+ * Idempotent; nothing to do is exit 0. */
122
+ export function unmountAllRestoresScript(): string {
123
+ return [
124
+ `awk '($3 ~ /^fuse/) && ($2 ~ /^\\/workspace\\// || $2 ~ /^${RESTORE_MOUNT_ROOT.replace(/\//g, "\\/")}\\//) {print length($2), $2}' /proc/mounts 2>/dev/null | sort -rn | cut -d' ' -f2- | while read -r _m; do fusermount3 -u "$_m" 2>/dev/null || umount -l "$_m" 2>/dev/null || true; done`,
125
+ `rm -rf ${RESTORE_MOUNT_ROOT}/* 2>/dev/null || true`,
126
+ `rm -f /var/backups/*.sqsh 2>/dev/null || true`,
127
+ `rm -rf /workspace/*.restore-* /workspace/*.extract-* 2>/dev/null || true`,
128
+ `true`,
129
+ ].join("\n");
130
+ }