@coreplane/switchboard 1.212.0 → 1.213.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/README.md +5 -3
  2. package/dist/assets/config/config.example.yaml +44 -0
  3. package/dist/assets/deploy/cloudflare/worker.ts +6 -0
  4. package/dist/assets/deploy/cloudflare-resident/Dockerfile +24 -0
  5. package/dist/assets/deploy/cloudflare-resident/worker.ts +353 -18
  6. package/dist/assets/deploy/cloudflare-resident/wrangler.template.jsonc +9 -1
  7. package/dist/assets/deploy/cloudflare-sandbox/Dockerfile +18 -0
  8. package/dist/assets/deploy/secrets.manifest.json +18 -0
  9. package/dist/assets/package-lock.json +5 -3
  10. package/dist/assets/package.json +2 -1
  11. package/dist/assets/source.json +3 -3
  12. package/dist/assets/src/agents/registry.ts +37 -0
  13. package/dist/assets/src/core/redact.ts +11 -1
  14. package/dist/assets/src/core/runEvents.ts +41 -1
  15. package/dist/assets/src/core/runFriction.ts +6 -4
  16. package/dist/assets/src/core/trace/workerTrace.ts +4 -0
  17. package/dist/assets/src/execution/residentRefresh.ts +151 -5
  18. package/dist/assets/web/dist/.vite/manifest.json +19 -19
  19. package/dist/assets/web/dist/assets/{ResidentDetailPage-CM5nWw-Z.js → ResidentDetailPage-CvwjhjlG.js} +1 -1
  20. package/dist/assets/web/dist/assets/{ResidentsIndexPage-BKBnuvp3.js → ResidentsIndexPage-DoN6XuNx.js} +1 -1
  21. package/dist/assets/web/dist/assets/RunRoutePage-FzLNexjF.js +12 -0
  22. package/dist/assets/web/dist/assets/{RunsIndexPage-w2jJP5xu.js → RunsIndexPage-DhSUbF9k.js} +1 -1
  23. package/dist/assets/web/dist/assets/{ScheduledPage-D3-k9DPz.js → ScheduledPage-Bq0Yg4nT.js} +1 -1
  24. package/dist/assets/web/dist/assets/{StatusDot-BmFHnV8m.js → StatusDot-D8Wwt1KC.js} +1 -1
  25. package/dist/assets/web/dist/assets/{Tooltip-DoThP2fW.js → Tooltip-DpDK7jWZ.js} +1 -1
  26. package/dist/assets/web/dist/assets/{dist-YrRKtxsS.js → dist-B7BVkB7x.js} +1 -1
  27. package/dist/assets/web/dist/assets/{main-D6nzMf0k.js → main-BFkEOy3K.js} +2 -2
  28. package/dist/assets/web/dist/assets/main-DCH3Mezs.css +1 -0
  29. package/dist/cli.js +2336 -328
  30. package/package.json +2 -1
  31. package/dist/assets/web/dist/assets/RunRoutePage-Bu4CwEzt.js +0 -12
  32. package/dist/assets/web/dist/assets/main-CuENKPdD.css +0 -1
@@ -114,6 +114,7 @@ import {
114
114
  recoverCapturedOutput,
115
115
  } from "../../src/execution/residentExecWrap.js";
116
116
  import { shellQuote } from "../../src/execution/shellQuote.js";
117
+ import { envFromRequest } from "../../src/execution/sandboxEnv.js";
117
118
  import {
118
119
  CREDENTIAL_EXPIRY_MARGIN_MS,
119
120
  shouldRefreshThreadCredentials,
@@ -145,9 +146,15 @@ import {
145
146
  checkoutUpdateCommand,
146
147
  classifyRefreshFailure,
147
148
  restoreFailureDisposition,
149
+ isRuntimeUnreachableSignal,
148
150
  killStaleBuildProcessesCommand,
149
151
  planRefresh,
150
152
  RUNTIME_REPLACEMENT_WORDING,
153
+ RUNTIME_UNREACHABLE_DOWN_AT,
154
+ runtimeUnreachableReason,
155
+ runtimeUnreachableRung,
156
+ SDK_CONNECT_TIMEOUT_MS,
157
+ SDK_RUNTIME_RECORD_KEY,
151
158
  judgeRestoreProgress,
152
159
  planWakeDepsBudget,
153
160
  RESTORE_MAX_MS,
@@ -508,8 +515,10 @@ const DEGRADED_STREAK_KEY = "resident:degradedStreak";
508
515
  * re-open the parked-degraded hole this fixes. A refresh step killed from
509
516
  * outside (`refresh-interrupted`, `classifyRefreshFailure`) is never recorded
510
517
  * as `degraded` at all: the instance throws it to the engine, whose retry
511
- * re-enters the step. */
512
- const NON_EVIDENCE_REASON = /^(?:stale-mid-flight|restore-interrupted):/;
518
+ * re-enters the step. A control port that did not answer
519
+ * (`runtime-unreachable: …`, item 64) is the third: no command ran, the
520
+ * ladder over its own persisted count owns the recovery. */
521
+ const NON_EVIDENCE_REASON = /^(?:stale-mid-flight|restore-interrupted|runtime-unreachable):/;
513
522
  /** When the disk-full recovery last stopped the container (docs/reference/specs/resident-repos.md item 54):
514
523
  * feeds `planDiskFullRecovery`'s cooldown so a working set that refills the
515
524
  * disk is named, not recycled in a loop. */
@@ -548,9 +557,11 @@ const ATTACH_MUTEX_WAIT_MS = 60_000;
548
557
  * an admin would, discarding the unusable snapshots and reprovisioning from
549
558
  * GitHub. Provision-failure downs never auto-rebuild — they would loop
550
559
  * against the same broken build. With the 10-minute cron, N=3 ≈ 30 minutes
551
- * down before the automatic escape hatch fires. */
560
+ * down before the automatic escape hatch fires. `runtime-unreachable` is the
561
+ * ladder's last rung (item 64): a recreated container did not answer either,
562
+ * so the rebuild — destroy plus reprovision — is the only exit left. */
552
563
  const AUTO_REBUILD_AFTER_STRIKES = 3;
553
- const REHYDRATION_FAILURE_RE = /^(r2-restore-failed|snapshot-stamp-mismatch|no-snapshot)/;
564
+ const REHYDRATION_FAILURE_RE = /^(r2-restore-failed|snapshot-stamp-mismatch|no-snapshot|runtime-unreachable)/;
554
565
 
555
566
  /** /exec budget: the shared 5-minute default (`BASH_TIMEOUT_MS`);
556
567
  * a caller may raise it per call via the body's `timeoutMs` up to the shared
@@ -714,6 +725,39 @@ function isRuntimeReplacement(err: unknown): boolean {
714
725
  function runtimeReplacedErr(err: RuntimeReplacedError): ThreadErr {
715
726
  return { error: err.message, status: 409, reason: "runtime-replaced" };
716
727
  }
728
+
729
+ /** The container's control port never answered: `exec` rejected with the
730
+ * DOMException of the SDK's connect abort (`DEFAULT_CONNECT_TIMEOUT_MS`,
731
+ * 30 s), raised inside its wake path — `RuntimeBootstrapProbe.probe` →
732
+ * `ContainerControlConnection.fetchUpgradeAttempt`, the WebSocket upgrade to
733
+ * port 3000 — before any process could start. Not a replacement (the runtime
734
+ * did not change; it is silent), not a command failure (nothing ran), not
735
+ * evidence about the repository. Carries the persisted consecutive count the
736
+ * ladder decides on (`runtimeUnreachableRung`, docs/reference/specs/resident-repos.md
737
+ * item 64); the message is the named reason, never the SDK's bare
738
+ * `The operation was aborted` — which is what the incident this names sat
739
+ * behind as `degraded(refresh-failed: The operation was aborted)` for forty
740
+ * minutes while nothing escalated. */
741
+ class RuntimeUnreachableError extends Error {
742
+ constructor(
743
+ readonly count: number,
744
+ readonly cause: unknown,
745
+ ) {
746
+ super(runtimeUnreachableReason(count));
747
+ this.name = "RuntimeUnreachableError";
748
+ }
749
+ }
750
+
751
+ /** Does this exec rejection mean the control port never answered? The pure
752
+ * signal (`isRuntimeUnreachableSignal`: the `AbortError` name, or the
753
+ * DOMException's message when a wrapper copied only that) over the error and
754
+ * its cause chain. Asked only AFTER `isRuntimeReplacement` — a replaced
755
+ * runtime is a different fact — and only of a spawn-phase error: a
756
+ * command's own output is a `StepError` with an exit code and never gets here. */
757
+ function isRuntimeUnreachable(err: unknown): boolean {
758
+ for (const link of selfAndCauses(err)) if (isRuntimeUnreachableSignal(link)) return true;
759
+ return false;
760
+ }
717
761
  /** Trailing slice of one string for an error reason. Command RESULTS are not
718
762
  * described here — `describeStepFailure` owns that, because choosing between
719
763
  * the two streams is what lost a diagnosis (residentStepReport.ts). */
@@ -1272,6 +1316,17 @@ const FACTS_KEY = "resident:facts";
1272
1316
  const SNAPSHOT_KEY = "resident:snapshot";
1273
1317
  const DEADLINE_AT_KEY = "resident:provisionDeadlineAt";
1274
1318
  const REBUILD_STRIKES_KEY = "resident:rebuildStrikes"; // watchdog auto-rebuild counter
1319
+ /** Consecutive connects the container's control port did not answer (item 64): the ladder's count. */
1320
+ const RUNTIME_UNREACHABLE_KEY = "resident:runtimeUnreachable";
1321
+
1322
+ /** The row under RUNTIME_UNREACHABLE_KEY: the count and both instants, so the
1323
+ * fleet watch sees how long the runtime has been silent. Cleared by any exec
1324
+ * whose process spawned. */
1325
+ interface RuntimeUnreachableRow {
1326
+ count: number;
1327
+ firstAt: string;
1328
+ lastAt: string;
1329
+ }
1275
1330
 
1276
1331
  /** Thread bindings live under their own prefix, keyed by threadKey. */
1277
1332
  const THREAD_KEY_PREFIX = "thread:";
@@ -1713,7 +1768,17 @@ export class ResidentDO extends Sandbox<Env> {
1713
1768
  try {
1714
1769
  proc = await createExtensionProcessSandbox(this).exec(argv as unknown as SandboxCommand, launch);
1715
1770
  } catch (err) {
1716
- if (!isRuntimeReplacement(err)) throw err;
1771
+ if (!isRuntimeReplacement(err)) {
1772
+ // The control port never answered the SDK's connect (its 30 s abort,
1773
+ // raised inside the wake path): no process started and nothing about
1774
+ // the repository is known. Count it in storage — the ladder of item 64
1775
+ // reads the count — and name it, so no reason ever carries the bare
1776
+ // `The operation was aborted`.
1777
+ if (isRuntimeUnreachable(err)) {
1778
+ throw new RuntimeUnreachableError((await this.noteRuntimeUnreachable()).count, err);
1779
+ }
1780
+ throw err;
1781
+ }
1717
1782
  // Forward-looking gate, structurally unreachable today: in the pinned SDK
1718
1783
  // (@cloudflare/sandbox@0.13.0-next.751.1) every `reason:"runtime_replaced"`
1719
1784
  // site hardcodes `retryable:false`, so a replacement currently always
@@ -1728,6 +1793,9 @@ export class ResidentDO extends Sandbox<Env> {
1728
1793
  );
1729
1794
  proc = await createExtensionProcessSandbox(this).exec(argv as unknown as SandboxCommand, launch);
1730
1795
  }
1796
+ // The spawn is the proof the control port answers: a persisted count of
1797
+ // unanswered connects ends here, whatever the command goes on to do.
1798
+ await this.clearRuntimeUnreachable();
1731
1799
  try {
1732
1800
  const out = await proc.output({ encoding: "utf8", timeout: timeout + 30_000 });
1733
1801
  // `truncated` is the SDK saying the process log stream was cut past its
@@ -2404,10 +2472,20 @@ export class ResidentDO extends Sandbox<Env> {
2404
2472
  [DEADLINE_AT_KEY]: systemClock() + provisioningTimeoutMs,
2405
2473
  });
2406
2474
  await this.ctx.storage.delete([FACTS_KEY, SNAPSHOT_KEY]); // defensive: no stale facts from a past life
2407
- this.deleteSchedules(PROVISIONING_CALLBACK);
2408
- this.deleteSchedules(PROVISION_RUN_CALLBACK);
2409
- await this.schedule(Math.max(1, Math.ceil(provisioningTimeoutMs / 1000)), PROVISIONING_CALLBACK, resource);
2410
- await this.schedule(1, PROVISION_RUN_CALLBACK, resource);
2475
+ try {
2476
+ this.deleteSchedules(PROVISIONING_CALLBACK);
2477
+ this.deleteSchedules(PROVISION_RUN_CALLBACK);
2478
+ await this.schedule(Math.max(1, Math.ceil(provisioningTimeoutMs / 1000)), PROVISIONING_CALLBACK, resource);
2479
+ await this.schedule(1, PROVISION_RUN_CALLBACK, resource);
2480
+ } catch (err) {
2481
+ // Nothing armed, so nothing may say `onboarding`: the row goes back to
2482
+ // what it was, the way the onboard route frees the registry slot. An
2483
+ // `onboarding` with no schedule behind it used to sit until the
2484
+ // watchdog's provision-timeout (seen live after an offboard in the same
2485
+ // isolate).
2486
+ await this.ctx.storage.delete([RESOURCE_KEY, STATE_KEY, REASON_KEY, UPDATED_KEY, DEADLINE_AT_KEY]);
2487
+ throw err;
2488
+ }
2411
2489
  return { state: "onboarding", reason: "" };
2412
2490
  }
2413
2491
 
@@ -3008,6 +3086,15 @@ export class ResidentDO extends Sandbox<Env> {
3008
3086
  * class: `disk-full: …`, never serviceable, and the one failure the
3009
3087
  * resident can act on itself (recoverFromDiskFull). */
3010
3088
  private async classifyCycleError(err: unknown): Promise<RefreshFailure> {
3089
+ // The control port never answered (item 64): the count decides, and a disk
3090
+ // probe would only cost another 30 s abort against the same silent port.
3091
+ if (err instanceof RuntimeUnreachableError) {
3092
+ return classifyRefreshFailure({
3093
+ step: "refresh",
3094
+ message: err.message,
3095
+ runtimeUnreachable: { count: err.count },
3096
+ });
3097
+ }
3011
3098
  return err instanceof StepError
3012
3099
  ? await this.classifyFailure(err.step, err.message)
3013
3100
  : await this.classifyFailure("refresh", errMsg(err));
@@ -3029,6 +3116,145 @@ export class ResidentDO extends Sandbox<Env> {
3029
3116
  if (failure.diskFull) await this.recoverFromDiskFull(failure.reason, selfInFlight);
3030
3117
  }
3031
3118
 
3119
+ // -- the runtime that never answers (docs/reference/specs/resident-repos.md item 64) ------
3120
+ //
3121
+ // Every `sandbox.exec` of the incident this section names rejected after
3122
+ // exactly 30 s with the SDK's connect abort: the WebSocket upgrade to the
3123
+ // container's control port was never answered, for forty minutes, while the
3124
+ // refresh instance recorded `degraded(refresh-failed: The operation was
3125
+ // aborted)` every bucket and nothing escalated — an admin `stop-container`
3126
+ // (a SIGTERM the runtime ignored) did not help either. The recovery is a
3127
+ // ladder over a persisted count of consecutive unanswered connects: re-arm,
3128
+ // stop, destroy and restore from the snapshot, then down with a reason the
3129
+ // watchdog's auto-rebuild strikes apply to. The count lives in storage
3130
+ // because the isolate does not: a Worker deploy or an eviction between
3131
+ // attempts would otherwise restart the ladder at one.
3132
+
3133
+ /** The persisted count of consecutive connects the control port did not
3134
+ * answer, or null while it answers. */
3135
+ private async runtimeUnreachableRow(): Promise<RuntimeUnreachableRow | null> {
3136
+ return (await this.ctx.storage.get<RuntimeUnreachableRow>(RUNTIME_UNREACHABLE_KEY)) ?? null;
3137
+ }
3138
+
3139
+ /** One more unanswered connect: the count up by one, `firstAt` kept, `lastAt`
3140
+ * now — and one log line naming the attempt (the SDK's own line is the bare
3141
+ * AbortError with a stack). */
3142
+ private async noteRuntimeUnreachable(): Promise<RuntimeUnreachableRow> {
3143
+ const prev = await this.runtimeUnreachableRow();
3144
+ const now = new Date(systemClock()).toISOString();
3145
+ const row: RuntimeUnreachableRow = { count: (prev?.count ?? 0) + 1, firstAt: prev?.firstAt ?? now, lastAt: now };
3146
+ await this.ctx.storage.put(RUNTIME_UNREACHABLE_KEY, row);
3147
+ this.runtimeUnreachableSeen = true;
3148
+ console.log(
3149
+ `runtime-unreachable: the control port did not answer within ${SDK_CONNECT_TIMEOUT_MS / 1000} s — attempt ${row.count} of ${RUNTIME_UNREACHABLE_DOWN_AT} (first at ${row.firstAt})`,
3150
+ );
3151
+ return row;
3152
+ }
3153
+
3154
+ /** Whether a row may exist, so the hot path pays one storage read per
3155
+ * isolate and a delete only for a row that is there. Storage stays the
3156
+ * truth; this only says whether it is worth asking. */
3157
+ private runtimeUnreachableSeen: boolean | undefined;
3158
+
3159
+ /** A spawned process is the proof the control port answers: the row goes,
3160
+ * and the log says the silence ended. */
3161
+ private async clearRuntimeUnreachable(): Promise<void> {
3162
+ if (this.runtimeUnreachableSeen === undefined) {
3163
+ this.runtimeUnreachableSeen = (await this.runtimeUnreachableRow()) !== null;
3164
+ }
3165
+ if (!this.runtimeUnreachableSeen) return;
3166
+ const row = await this.runtimeUnreachableRow();
3167
+ await this.ctx.storage.delete(RUNTIME_UNREACHABLE_KEY);
3168
+ this.runtimeUnreachableSeen = false;
3169
+ if (row) {
3170
+ console.log(
3171
+ `runtime-unreachable: cleared — the control port answered again after ${row.count} unanswered attempt(s) since ${row.firstAt}`,
3172
+ );
3173
+ }
3174
+ }
3175
+
3176
+ /** The ladder (`runtimeUnreachableRung`) over the count `run()` persisted,
3177
+ * applied where a refresh step's exec found the port silent. Every rung
3178
+ * records its reason (`lastRefreshError` and the state) and logs one line
3179
+ * naming the rung and the count; the first three then throw the step back
3180
+ * to the engine, whose retry re-enters the same idempotent method thirty
3181
+ * seconds on, doubling — within one instance's six attempts the ladder runs
3182
+ * from the first unanswered connect to `down`, and a count that outlives the
3183
+ * instance carries into the next bucket's. The state is `degraded` under
3184
+ * every rung but the last: a `restoring` marker with no restore running
3185
+ * would hold the cron's instance creation off until the stale bound, so the
3186
+ * retry's wake path flips `restoring` itself when it starts the restore. */
3187
+ private async escalateRuntimeUnreachable(
3188
+ instance: string,
3189
+ step: string,
3190
+ err: RuntimeUnreachableError,
3191
+ ): Promise<{ status: "failed"; reason: string }> {
3192
+ const rung = runtimeUnreachableRung(err.count);
3193
+ const reason = runtimeUnreachableReason(err.count, rung);
3194
+ console.log(
3195
+ `refresh instance ${instance}: ${step} runtime-unreachable — rung ${rung} at attempt ${err.count} of ${RUNTIME_UNREACHABLE_DOWN_AT}`,
3196
+ );
3197
+ await this.recordRefreshError(reason);
3198
+ switch (rung) {
3199
+ case "re-arm":
3200
+ await this.setResidentState("degraded", reason);
3201
+ throw err;
3202
+ case "stop":
3203
+ // SIGTERM (`stop()` signals and returns; it cannot kill), the
3204
+ // incarnation swapped: a runtime that still honours signals restarts
3205
+ // under the retry's exec on a fresh disk, and the wake path restores.
3206
+ this.swapIncarnation(); // deliberate incarnation swap
3207
+ await this.stop().catch((stopErr) => console.log(`runtime-unreachable: stop failed: ${errMsg(stopErr)}`));
3208
+ await this.setResidentState("degraded", reason);
3209
+ throw err;
3210
+ case "recreate":
3211
+ await this.recreateContainer(reason);
3212
+ throw err;
3213
+ case "down":
3214
+ // A fresh VM did not answer either. Down with a strike-eligible reason
3215
+ // (REHYDRATION_FAILURE_RE) — the watchdog rebuilds after its passes —
3216
+ // and the VM destroyed, so the rebuild's provisioning starts on a new one.
3217
+ this.swapIncarnation(); // deliberate incarnation swap
3218
+ await this.destroy().catch((destroyErr) =>
3219
+ console.log(`runtime-unreachable: destroy failed: ${errMsg(destroyErr)}`),
3220
+ );
3221
+ await this.clearInstanceLease(instance);
3222
+ return { status: "failed", reason: (await this.goDown(reason)).reason };
3223
+ }
3224
+ }
3225
+
3226
+ /** Destroy the VM and keep everything else: the snapshots, the entry
3227
+ * backups, the registry record, the bindings. `destroy()` is the SDK's
3228
+ * SIGKILL of the whole container (`ctx.container.destroy()`), where `stop()`
3229
+ * is a SIGTERM the runtime may ignore — a control server that no longer
3230
+ * answers its port may not answer signals either, which is what the
3231
+ * incident's admin `stop-container` showed. The disk goes with the VM; the
3232
+ * next exec's wake path finds no runtime, flips `restoring` and restores
3233
+ * mirror, checkout and deps from R2 — the cheap recovery (minutes), where a
3234
+ * rebuild (destroy plus reprovision from the code host) is the expensive
3235
+ * one. The state stays `degraded` with the reason naming the pending
3236
+ * restore, for the reason `escalateRuntimeUnreachable` gives. */
3237
+ private async recreateContainer(reason: string): Promise<void> {
3238
+ console.log(
3239
+ `runtime-unreachable: destroying the container — snapshots kept; the next exec restores from R2 (${reason.slice(0, 200)})`,
3240
+ );
3241
+ this.swapIncarnation(); // deliberate incarnation swap
3242
+ await this.forgetRuntimeIdentity();
3243
+ await this.destroy().catch((err) => console.log(`runtime-unreachable: destroy failed: ${errMsg(err)}`));
3244
+ await this.setResidentState("degraded", reason);
3245
+ }
3246
+
3247
+ /** Forget the SDK's stored runtime identity (`SDK_RUNTIME_RECORD_KEY`)
3248
+ * before a destroy. The SDK's own `stop()` and `destroy()` delete it
3249
+ * (`invalidate`) — this is the guard for the path where they do not get
3250
+ * that far, and it makes the destroy prompt: with no identity stored the
3251
+ * SDK skips the runtime cleanup it would otherwise attempt against the
3252
+ * silent port. Never a recovery on its own: the incident's `stop-container`
3253
+ * had already deleted the record and the next connect aborted the same way. */
3254
+ private async forgetRuntimeIdentity(): Promise<void> {
3255
+ await this.ctx.storage.delete(SDK_RUNTIME_RECORD_KEY);
3256
+ }
3257
+
3032
3258
  // -- the refresh cycle as a Workflow instance (item 7) --------------------------
3033
3259
  //
3034
3260
  // `ResidentRefresh` (the Workflow entrypoint, refresh.ts) calls these
@@ -3224,6 +3450,20 @@ export class ResidentDO extends Sandbox<Env> {
3224
3450
  console.log(`refresh instance ${instance}: ${step} ${err.message}`);
3225
3451
  throw err;
3226
3452
  }
3453
+ if (err instanceof RuntimeUnreachableError) {
3454
+ // The control port never answered (item 64): the ladder decides — the
3455
+ // step is thrown back for the engine's retry under the first three
3456
+ // rungs, or the resident is down under the last.
3457
+ let result: { status: "failed"; reason: string };
3458
+ try {
3459
+ result = await this.escalateRuntimeUnreachable(instance, step, err);
3460
+ } catch (rethrown) {
3461
+ outcome = `runtime-unreachable (attempt ${err.count} of ${RUNTIME_UNREACHABLE_DOWN_AT}, rung ${runtimeUnreachableRung(err.count)}) — the engine retries`;
3462
+ throw rethrown;
3463
+ }
3464
+ outcome = `failed (${result.reason})`;
3465
+ return { ...result, startedAt, trace: trace.steps() };
3466
+ }
3227
3467
  const failure = await this.classifyCycleError(err);
3228
3468
  if (failure.interrupted) {
3229
3469
  outcome = `interrupted (${failure.reason}) — the engine retries`;
@@ -3420,6 +3660,10 @@ export class ResidentDO extends Sandbox<Env> {
3420
3660
  }): Promise<InstanceStepAnswer<{ result: { evicted: string[]; kept: number } | null; error: string | null }>> {
3421
3661
  return this.runInstanceStep(input.instance, "sweep", async (cycle) => {
3422
3662
  cycle.count();
3663
+ // A runtime that does not answer has nothing to sweep, and every probe
3664
+ // would cost the SDK's 30 s abort (item 64); the fetch step already
3665
+ // recorded the verdict this instance.
3666
+ if (await this.runtimeUnreachableRow()) return { status: "stopped", why: "runtime-unreachable" };
3423
3667
  return this.housekeeping("sweep", () => this.sweepWorktrees(input.resource));
3424
3668
  });
3425
3669
  }
@@ -3433,6 +3677,8 @@ export class ResidentDO extends Sandbox<Env> {
3433
3677
  }): Promise<InstanceStepAnswer<{ result: { measured: boolean } | null; error: string | null }>> {
3434
3678
  return this.runInstanceStep(input.instance, "measure", async (cycle) => {
3435
3679
  cycle.count();
3680
+ // Same gate as the sweep: a silent control port cannot answer a `df`.
3681
+ if (await this.runtimeUnreachableRow()) return { status: "stopped", why: "runtime-unreachable" };
3436
3682
  return this.housekeeping("measure", async () => ({ measured: (await this.measureDisk()) !== null }));
3437
3683
  });
3438
3684
  }
@@ -3942,8 +4188,13 @@ export class ResidentDO extends Sandbox<Env> {
3942
4188
  timeoutMs: number,
3943
4189
  capBytes?: number,
3944
4190
  capFiles?: { out: string; err: string },
4191
+ env?: Record<string, string>,
3945
4192
  ): Promise<{ stdout: string; stderr: string; exitCode: number; timedOut: boolean; truncated?: boolean }> {
3946
- const injected = { GIT_TERMINAL_PROMPT: "0" };
4193
+ // The caller's variables (an /exec body's `env`, docs/reference/specs/
4194
+ // harness-pi.md item 4) under the Worker's own: a caller never overrides
4195
+ // what the Worker injects. `su` without `-` keeps this environment for the
4196
+ // thread user's shell.
4197
+ const injected = { ...(env ?? {}), GIT_TERMINAL_PROMPT: "0" };
3947
4198
  validateEnvNames(injected);
3948
4199
  const body = capBytes
3949
4200
  ? capWrappedCommand(worktreePath, command, capBytes, capFiles)
@@ -3965,10 +4216,11 @@ export class ResidentDO extends Sandbox<Env> {
3965
4216
  command: string,
3966
4217
  timeoutMs: number,
3967
4218
  charCap: number,
4219
+ env?: Record<string, string>,
3968
4220
  ): Promise<{ stdout: string; stderr: string; exitCode: number; timedOut: boolean; truncated?: boolean }> {
3969
4221
  const capBytes = capBytesFor(charCap);
3970
4222
  const files = execCapFiles();
3971
- const r = await this.threadRun(user, worktreePath, command, timeoutMs, capBytes, files);
4223
+ const r = await this.threadRun(user, worktreePath, command, timeoutMs, capBytes, files, env);
3972
4224
  if (!r.timedOut) return r;
3973
4225
  try {
3974
4226
  const rec = await this.threadRun(
@@ -5130,12 +5382,13 @@ export class ResidentDO extends Sandbox<Env> {
5130
5382
  command: string,
5131
5383
  timeoutMs: number,
5132
5384
  traceparent?: string,
5385
+ env?: Record<string, string>,
5133
5386
  ): Promise<{ stdout: string; stderr: string; exitCode: number; truncated: boolean } | ThreadErr> {
5134
5387
  const queuedAt = systemClock();
5135
5388
  let startedAt = queuedAt;
5136
5389
  const res = await this.withThreadBusy(threadKey, () => {
5137
5390
  startedAt = systemClock();
5138
- return this.execThreadImpl(threadKey, command, timeoutMs);
5391
+ return this.execThreadImpl(threadKey, command, timeoutMs, env);
5139
5392
  });
5140
5393
  // The command as the resident's own `resident.exec` root (docs/reference/specs/tracing.md
5141
5394
  // item 22): started when the command did, the wait for the thread's turn an attr.
@@ -5150,6 +5403,7 @@ export class ResidentDO extends Sandbox<Env> {
5150
5403
  threadKey: string,
5151
5404
  command: string,
5152
5405
  timeoutMs: number,
5406
+ env?: Record<string, string>,
5153
5407
  ): Promise<{ stdout: string; stderr: string; exitCode: number; truncated: boolean } | ThreadErr> {
5154
5408
  const pre = await this.threadPreflight(threadKey);
5155
5409
  if ("error" in pre) return pre;
@@ -5167,7 +5421,7 @@ export class ResidentDO extends Sandbox<Env> {
5167
5421
 
5168
5422
  let r: Awaited<ReturnType<ResidentDO["threadRun"]>>;
5169
5423
  try {
5170
- r = await this.threadRunCapped(binding.user, binding.worktreePath, command, timeoutMs, EXEC_OUTPUT_CAP);
5424
+ r = await this.threadRunCapped(binding.user, binding.worktreePath, command, timeoutMs, EXEC_OUTPUT_CAP, env);
5171
5425
  } catch (err) {
5172
5426
  if (err instanceof RuntimeReplacedError) return runtimeReplacedErr(err);
5173
5427
  throw err;
@@ -6011,10 +6265,12 @@ export class ResidentDO extends Sandbox<Env> {
6011
6265
  inFlightKey("hydration"),
6012
6266
  LIFECYCLE_KEY,
6013
6267
  REFRESH_INSTANCE_KEY,
6268
+ RUNTIME_UNREACHABLE_KEY,
6014
6269
  ]);
6015
6270
  const facts = map.get(FACTS_KEY) as RepoFacts | undefined;
6016
6271
  const snap = map.get(SNAPSHOT_KEY) as SnapshotRecord | undefined;
6017
6272
  const disk = (map.get(DISK_KEY) as DiskSample | undefined) ?? null;
6273
+ const unreachable = (map.get(RUNTIME_UNREACHABLE_KEY) as RuntimeUnreachableRow | undefined) ?? null;
6018
6274
  const refreshRow = (map.get(REFRESH_INSTANCE_KEY) as RefreshInstanceRow | undefined) ?? {
6019
6275
  instance: null,
6020
6276
  skipped: null,
@@ -6097,6 +6353,9 @@ export class ResidentDO extends Sandbox<Env> {
6097
6353
  // Item 55: the last disk sample (`residentDiskBudget.ts` DiskSample), or
6098
6354
  // null before the first measurement of this incarnation.
6099
6355
  disk,
6356
+ // Item 64: consecutive connects the control port did not answer, with
6357
+ // the rung that count is on; null while the port answers.
6358
+ runtimeUnreachable: unreachable ? { ...unreachable, rung: runtimeUnreachableRung(unreachable.count) } : null,
6100
6359
  };
6101
6360
  }
6102
6361
 
@@ -6135,6 +6394,50 @@ export class ResidentDO extends Sandbox<Env> {
6135
6394
  }
6136
6395
  }
6137
6396
 
6397
+ /** Admin `recreate-container` (item 13; item 64's rung 3 on demand): destroy
6398
+ * the VM, keep every snapshot, and start the restore now — the operator's
6399
+ * recovery for a resident whose runtime never answers, minutes where the
6400
+ * rebuild is half an hour. Refused while the engine owns the state
6401
+ * (onboarding/refreshing/restoring — two cycles must never race one disk)
6402
+ * and on a `down` resident, whose one exit is `/rebuild`. The restore runs
6403
+ * in the background through the ordinary wake path (`ensureHydrated`:
6404
+ * `restoring`, mirror, checkout, deps, `warm`, `lastRestore`); the caller
6405
+ * polls `/debug info`. A failure the wake path did not record itself is
6406
+ * recorded here, so the marker never strands `restoring`. */
6407
+ async debugRecreateContainer(): Promise<
6408
+ { recreated: true; restoreStartedAt: string } | { recreated: false; error: string; status: number }
6409
+ > {
6410
+ const from = await this.getStatus();
6411
+ if (from.state === "onboarding" || from.state === "refreshing" || from.state === "restoring") {
6412
+ return {
6413
+ recreated: false,
6414
+ status: 409,
6415
+ error: `recreate-refused: the engine is mid-flight (state ${from.state}) — retry once it settles (warm/degraded)`,
6416
+ };
6417
+ }
6418
+ if (from.state === "down") {
6419
+ return {
6420
+ recreated: false,
6421
+ status: 409,
6422
+ error: `recreate-refused: the resident is down (${from.reason}) — POST /rebuild is its exit`,
6423
+ };
6424
+ }
6425
+ await this.recreateContainer(
6426
+ "runtime-unreachable: the container was recreated by an operator (recreate-container), snapshots kept — the restore from the snapshot is starting",
6427
+ );
6428
+ const restoreStartedAt = new Date(systemClock()).toISOString();
6429
+ this.ctx.waitUntil(
6430
+ this.ensureHydrated().catch(async (err) => {
6431
+ if (err instanceof ResidentDownError) return; // the wake path recorded it
6432
+ const failure = await this.classifyCycleError(err);
6433
+ console.log(`recreate-container: the restore failed — ${failure.reason.slice(0, 400)}`);
6434
+ await this.recordRefreshError(failure.reason);
6435
+ if ((await this.getStatus()).state === "restoring") await this.setResidentState("degraded", failure.reason);
6436
+ }),
6437
+ );
6438
+ return { recreated: true, restoreStartedAt };
6439
+ }
6440
+
6138
6441
  /** Fault injection for the watchdog's auto-rebuild path: persist
6139
6442
  * `down` with a rehydration-flavored reason (what a real goDown does), so
6140
6443
  * repeated watchdog passes can strike it up to the auto-rebuild without
@@ -6222,6 +6525,16 @@ export class ResidentDO extends Sandbox<Env> {
6222
6525
  }
6223
6526
  }
6224
6527
  await this.ctx.storage.delete(REBUILD_STRIKES_KEY);
6528
+ // From scratch means a fresh container too: the SDK's runtime identity
6529
+ // forgotten and the VM destroyed (SIGKILL, a fresh disk) before
6530
+ // provisioning clones onto it. A rebuild that reprovisioned onto the
6531
+ // running container inherited its wedged runtime once — every exec of the
6532
+ // new provisioning met the same unanswered control port (item 64).
6533
+ this.swapIncarnation(); // deliberate incarnation swap
6534
+ await this.forgetRuntimeIdentity();
6535
+ await this.destroy().catch((err) =>
6536
+ console.log(`rebuild: destroy failed (provisioning starts anyway): ${errMsg(err)}`),
6537
+ );
6225
6538
  await this.initResident(resource, provisioningTimeoutMs);
6226
6539
  return { ...plan, backupObjectsDeleted, state: "onboarding" as const };
6227
6540
  }
@@ -6293,10 +6606,17 @@ export class ResidentDO extends Sandbox<Env> {
6293
6606
  errors.push(`destroy failed: ${errMsg(err)}`);
6294
6607
  }
6295
6608
  // Retired DO: clear the alarm the Container base may have armed for its
6296
- // schedules, then wipe storage so nothing ever wakes this object again.
6609
+ // schedules, then delete every stored key — ours and the SDK's (its runtime
6610
+ // identity among them) — so nothing ever wakes this object again. Keys, not
6611
+ // `deleteAll()`: on a SQLite-backed object that also drops the SDK's
6612
+ // `container_schedules` table, which only its constructor creates, and an
6613
+ // onboard served by this same isolate then fails arming with
6614
+ // `no such table` after writing `onboarding` (seen live). The table stays,
6615
+ // empty — its rows are the two schedules cancelled above.
6297
6616
  this.swapIncarnation(); // retired object, retired memos
6298
6617
  await this.ctx.storage.deleteAlarm();
6299
- await this.ctx.storage.deleteAll();
6618
+ const keys = [...(await this.ctx.storage.list()).keys()];
6619
+ for (let i = 0; i < keys.length; i += 128) await this.ctx.storage.delete(keys.slice(i, i + 128));
6300
6620
  return { schedulesCancelled: true, containerStopped, storageCleared: true, backupObjectsDeleted, errors };
6301
6621
  }
6302
6622
  }
@@ -6735,7 +7055,10 @@ async function handleOnboard(env: Env, body: Record<string, unknown>): Promise<R
6735
7055
  {
6736
7056
  error:
6737
7057
  `not-in-installation: the GitHub App cannot mint a token scoped to ${resource.resource} — ` +
6738
- `install the App on the repository first (${errMsg(err)})`,
7058
+ `the repository is not in the App installation's repository list, or does not exist under that ` +
7059
+ `exact name (GitHub's token API answers the same 422 for both). An org admin adds it under the ` +
7060
+ `App's installation settings (Settings → GitHub Apps → Configure → Repository access), ` +
7061
+ `then retry (${errMsg(err)})`,
6739
7062
  },
6740
7063
  403,
6741
7064
  );
@@ -7183,7 +7506,12 @@ async function handleExec(env: Env, body: Record<string, unknown>, traceparent?:
7183
7506
  // a string) runs at the 5-minute default. A clamp, not a 400: an out-of-range
7184
7507
  // ask still runs, at the nearest bound.
7185
7508
  const timeoutMs = clampBashTimeout(body.timeoutMs);
7186
- return streamThreadExec(ctx.stub.execThread(ctx.threadKey, body.command, timeoutMs, traceparent));
7509
+ // A caller's extra environment for this one command (docs/reference/specs/
7510
+ // harness-pi.md item 4) — the run bearer the pi harness hands its process —
7511
+ // read from the body alone through the one validated reader the sandbox
7512
+ // Worker uses, and handed to the exec's env option, never onto the command.
7513
+ const execEnv = envFromRequest({ body });
7514
+ return streamThreadExec(ctx.stub.execThread(ctx.threadKey, body.command, timeoutMs, traceparent, execEnv));
7187
7515
  }
7188
7516
 
7189
7517
  /** Stream one pending result with the thread-sandbox Worker's heartbeat
@@ -7423,6 +7751,13 @@ async function handleDebug(env: Env, body: Record<string, unknown>): Promise<Res
7423
7751
  });
7424
7752
  case "stop-container":
7425
7753
  return json(await stub.debugStopContainer());
7754
+ case "recreate-container": {
7755
+ // Item 64's rung 3 on demand: destroy the VM, keep the snapshots, start
7756
+ // the restore now. `in` narrowing, as handleRebuild (the RPC stub's
7757
+ // Disposable intersection defeats the boolean discriminant).
7758
+ const r = await stub.debugRecreateContainer();
7759
+ return "error" in r ? json({ error: r.error }, r.status) : json(r, 202);
7760
+ }
7426
7761
  case "force-onboarding":
7427
7762
  return json(await stub.debugForceOnboarding());
7428
7763
  case "force-down": {
@@ -7468,7 +7803,7 @@ async function handleDebug(env: Env, body: Record<string, unknown>): Promise<Res
7468
7803
  default:
7469
7804
  return json(
7470
7805
  {
7471
- error: `unknown op ${JSON.stringify(op)} (ops: info, schedules, refresh-now, stop-container, force-onboarding, force-down, mint-token, run-watchdog, set-test-overrides, threads, sweep-now, reclaim-now, measure-disk, purge-bindings, backdate-thread, lifecycle)`,
7806
+ error: `unknown op ${JSON.stringify(op)} (ops: info, schedules, refresh-now, stop-container, recreate-container, force-onboarding, force-down, mint-token, run-watchdog, set-test-overrides, threads, sweep-now, reclaim-now, measure-disk, purge-bindings, backdate-thread, lifecycle)`,
7472
7807
  },
7473
7808
  400,
7474
7809
  );
@@ -30,7 +30,15 @@
30
30
  // binding's bucket below — the SDK signs URLs for this name, the offboard
31
31
  // sweep deletes through the binding. Public config, not secrets.
32
32
  "CLOUDFLARE_ACCOUNT_ID": "{{account}}",
33
- "BACKUP_BUCKET_NAME": "{{script}}-cache"
33
+ "BACKUP_BUCKET_NAME": "{{script}}-cache",
34
+ // The wake budget (docs/reference/specs/resident-repos.md item 64): how long
35
+ // the SDK's physical start waits for the container's control port to accept
36
+ // a request before the first connect — WAKE_PORT_READY_MS in
37
+ // src/execution/residentRefresh.ts, 3 min in place of the SDK's 90 s, so a
38
+ // slow boot (a 2.65 GB image on a cold host) is not read as an unreachable
39
+ // runtime. The env name and its bounds (10 s to 600 s) are the SDK's; a test
40
+ // pins this value to the constant.
41
+ "SANDBOX_PORT_TIMEOUT_MS": "180000"
34
42
  },
35
43
  "containers": [
36
44
  {
@@ -136,3 +136,21 @@ RUN apt-get update \
136
136
  && playwright --version | grep -qx 'Version 1.63.0' \
137
137
  && playwright screenshot --viewport-size=640,480 'data:text/html,<h1>ok</h1>' /tmp/ok.png && test -s /tmp/ok.png \
138
138
  && rm -f /tmp/proof.mp4 /tmp/proof_*.png /tmp/ok.png
139
+
140
+ # pi — the coding harness the bot can start INSIDE this container in place of
141
+ # its native turn loop (docs/reference/specs/harness-pi.md): `pi --mode rpc`,
142
+ # one process per run in the thread's worktree, driven by the bot over its
143
+ # JSONL protocol, with the run's model-proxy bearer as its only key (never a
144
+ # model key: the bearer buys calls through the bot, docs/reference/specs/
145
+ # model-proxy.md). Installed globally with this image's Node — pi needs
146
+ # >= 22.19.0 and the image ships 24 — so every user finds it on PATH, at an
147
+ # EXACT pin held by src/deploy/imagePiHarness.test.ts (the pnpm lesson: a
148
+ # floating tag would move the harness's protocol with the build date, and a
149
+ # pi minor changes RPC events and extension hooks). Proven by the layer, so
150
+ # the BUILD fails, not a run: the pi on PATH answers the pin, and its own help
151
+ # names the RPC mode the harness drives. Dark until a deployment sets
152
+ # `harness: { coding: pi }`: nothing here starts pi on its own.
153
+ RUN npm install -g @earendil-works/pi-coding-agent@0.85.1 \
154
+ && npm cache clean --force \
155
+ && pi --version | grep -qx '0.85.1' \
156
+ && pi --help | grep -q -- '--mode <mode>'
@@ -75,6 +75,24 @@
75
75
  "optional": true,
76
76
  "note": "The secret half of R2_ACCESS_KEY_ID (same token). Both or neither."
77
77
  },
78
+ {
79
+ "name": "ARTIFACTS_R2_ACCESS_KEY_ID",
80
+ "workers": ["bot"],
81
+ "optional": true,
82
+ "note": "R2 API token (S3 access key id) scoped Object Read & Write to the artifacts bucket ONLY (`artifacts.r2.bucket` in config.yaml; docs/reference/specs/execution.md item 20): the bot signs presigned PUT/GET URLs and HEADs objects with it; it never carries a file. Optional: absent with no `artifacts:` section means no store. Created in the Cloudflare dashboard (R2 → Manage API tokens); rotate = new token, `deploy secrets bot`, `deploy restart`."
83
+ },
84
+ {
85
+ "name": "ARTIFACTS_R2_SECRET_ACCESS_KEY",
86
+ "workers": ["bot"],
87
+ "optional": true,
88
+ "note": "The secret half of ARTIFACTS_R2_ACCESS_KEY_ID (same token). Both or neither."
89
+ },
90
+ {
91
+ "name": "ARTIFACTS_COPY_TOKEN",
92
+ "workers": ["bot"],
93
+ "optional": true,
94
+ "note": "Shared bearer between the bot and its own Worker's `POST /artifacts/copy` route, which streams an inbound Slack file into the artifacts bucket (the Worker holds the R2 binding and the Slack token; the bot only asks). Self-minted (`openssl rand -hex 32`); the Worker checks it, the container presents it. Required whenever `artifacts:` is configured — the bot fails fast at startup without it."
95
+ },
78
96
  {
79
97
  "name": "MCP_CREDENTIAL_KEY",
80
98
  "workers": ["bot"],