@coreplane/switchboard 1.212.0 → 1.213.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -3
- package/dist/assets/config/config.example.yaml +44 -0
- package/dist/assets/deploy/cloudflare/worker.ts +6 -0
- package/dist/assets/deploy/cloudflare-resident/Dockerfile +24 -0
- package/dist/assets/deploy/cloudflare-resident/worker.ts +353 -18
- package/dist/assets/deploy/cloudflare-resident/wrangler.template.jsonc +9 -1
- package/dist/assets/deploy/cloudflare-sandbox/Dockerfile +18 -0
- package/dist/assets/deploy/secrets.manifest.json +18 -0
- package/dist/assets/package-lock.json +5 -3
- package/dist/assets/package.json +2 -1
- package/dist/assets/source.json +3 -3
- package/dist/assets/src/agents/registry.ts +37 -0
- package/dist/assets/src/core/redact.ts +11 -1
- package/dist/assets/src/core/runEvents.ts +41 -1
- package/dist/assets/src/core/runFriction.ts +6 -4
- package/dist/assets/src/core/trace/workerTrace.ts +4 -0
- package/dist/assets/src/execution/residentRefresh.ts +151 -5
- package/dist/assets/web/dist/.vite/manifest.json +19 -19
- package/dist/assets/web/dist/assets/{ResidentDetailPage-CM5nWw-Z.js → ResidentDetailPage-CvwjhjlG.js} +1 -1
- package/dist/assets/web/dist/assets/{ResidentsIndexPage-BKBnuvp3.js → ResidentsIndexPage-DoN6XuNx.js} +1 -1
- package/dist/assets/web/dist/assets/RunRoutePage-FzLNexjF.js +12 -0
- package/dist/assets/web/dist/assets/{RunsIndexPage-w2jJP5xu.js → RunsIndexPage-DhSUbF9k.js} +1 -1
- package/dist/assets/web/dist/assets/{ScheduledPage-D3-k9DPz.js → ScheduledPage-Bq0Yg4nT.js} +1 -1
- package/dist/assets/web/dist/assets/{StatusDot-BmFHnV8m.js → StatusDot-D8Wwt1KC.js} +1 -1
- package/dist/assets/web/dist/assets/{Tooltip-DoThP2fW.js → Tooltip-DpDK7jWZ.js} +1 -1
- package/dist/assets/web/dist/assets/{dist-YrRKtxsS.js → dist-B7BVkB7x.js} +1 -1
- package/dist/assets/web/dist/assets/{main-D6nzMf0k.js → main-BFkEOy3K.js} +2 -2
- package/dist/assets/web/dist/assets/main-DCH3Mezs.css +1 -0
- package/dist/cli.js +2336 -328
- package/package.json +2 -1
- package/dist/assets/web/dist/assets/RunRoutePage-Bu4CwEzt.js +0 -12
- package/dist/assets/web/dist/assets/main-CuENKPdD.css +0 -1
|
@@ -114,6 +114,7 @@ import {
|
|
|
114
114
|
recoverCapturedOutput,
|
|
115
115
|
} from "../../src/execution/residentExecWrap.js";
|
|
116
116
|
import { shellQuote } from "../../src/execution/shellQuote.js";
|
|
117
|
+
import { envFromRequest } from "../../src/execution/sandboxEnv.js";
|
|
117
118
|
import {
|
|
118
119
|
CREDENTIAL_EXPIRY_MARGIN_MS,
|
|
119
120
|
shouldRefreshThreadCredentials,
|
|
@@ -145,9 +146,15 @@ import {
|
|
|
145
146
|
checkoutUpdateCommand,
|
|
146
147
|
classifyRefreshFailure,
|
|
147
148
|
restoreFailureDisposition,
|
|
149
|
+
isRuntimeUnreachableSignal,
|
|
148
150
|
killStaleBuildProcessesCommand,
|
|
149
151
|
planRefresh,
|
|
150
152
|
RUNTIME_REPLACEMENT_WORDING,
|
|
153
|
+
RUNTIME_UNREACHABLE_DOWN_AT,
|
|
154
|
+
runtimeUnreachableReason,
|
|
155
|
+
runtimeUnreachableRung,
|
|
156
|
+
SDK_CONNECT_TIMEOUT_MS,
|
|
157
|
+
SDK_RUNTIME_RECORD_KEY,
|
|
151
158
|
judgeRestoreProgress,
|
|
152
159
|
planWakeDepsBudget,
|
|
153
160
|
RESTORE_MAX_MS,
|
|
@@ -508,8 +515,10 @@ const DEGRADED_STREAK_KEY = "resident:degradedStreak";
|
|
|
508
515
|
* re-open the parked-degraded hole this fixes. A refresh step killed from
|
|
509
516
|
* outside (`refresh-interrupted`, `classifyRefreshFailure`) is never recorded
|
|
510
517
|
* as `degraded` at all: the instance throws it to the engine, whose retry
|
|
511
|
-
* re-enters the step.
|
|
512
|
-
|
|
518
|
+
* re-enters the step. A control port that did not answer
|
|
519
|
+
* (`runtime-unreachable: …`, item 64) is the third: no command ran, the
|
|
520
|
+
* ladder over its own persisted count owns the recovery. */
|
|
521
|
+
const NON_EVIDENCE_REASON = /^(?:stale-mid-flight|restore-interrupted|runtime-unreachable):/;
|
|
513
522
|
/** When the disk-full recovery last stopped the container (docs/reference/specs/resident-repos.md item 54):
|
|
514
523
|
* feeds `planDiskFullRecovery`'s cooldown so a working set that refills the
|
|
515
524
|
* disk is named, not recycled in a loop. */
|
|
@@ -548,9 +557,11 @@ const ATTACH_MUTEX_WAIT_MS = 60_000;
|
|
|
548
557
|
* an admin would, discarding the unusable snapshots and reprovisioning from
|
|
549
558
|
* GitHub. Provision-failure downs never auto-rebuild — they would loop
|
|
550
559
|
* against the same broken build. With the 10-minute cron, N=3 ≈ 30 minutes
|
|
551
|
-
* down before the automatic escape hatch fires.
|
|
560
|
+
* down before the automatic escape hatch fires. `runtime-unreachable` is the
|
|
561
|
+
* ladder's last rung (item 64): a recreated container did not answer either,
|
|
562
|
+
* so the rebuild — destroy plus reprovision — is the only exit left. */
|
|
552
563
|
const AUTO_REBUILD_AFTER_STRIKES = 3;
|
|
553
|
-
const REHYDRATION_FAILURE_RE = /^(r2-restore-failed|snapshot-stamp-mismatch|no-snapshot)/;
|
|
564
|
+
const REHYDRATION_FAILURE_RE = /^(r2-restore-failed|snapshot-stamp-mismatch|no-snapshot|runtime-unreachable)/;
|
|
554
565
|
|
|
555
566
|
/** /exec budget: the shared 5-minute default (`BASH_TIMEOUT_MS`);
|
|
556
567
|
* a caller may raise it per call via the body's `timeoutMs` up to the shared
|
|
@@ -714,6 +725,39 @@ function isRuntimeReplacement(err: unknown): boolean {
|
|
|
714
725
|
function runtimeReplacedErr(err: RuntimeReplacedError): ThreadErr {
|
|
715
726
|
return { error: err.message, status: 409, reason: "runtime-replaced" };
|
|
716
727
|
}
|
|
728
|
+
|
|
729
|
+
/** The container's control port never answered: `exec` rejected with the
|
|
730
|
+
* DOMException of the SDK's connect abort (`DEFAULT_CONNECT_TIMEOUT_MS`,
|
|
731
|
+
* 30 s), raised inside its wake path — `RuntimeBootstrapProbe.probe` →
|
|
732
|
+
* `ContainerControlConnection.fetchUpgradeAttempt`, the WebSocket upgrade to
|
|
733
|
+
* port 3000 — before any process could start. Not a replacement (the runtime
|
|
734
|
+
* did not change; it is silent), not a command failure (nothing ran), not
|
|
735
|
+
* evidence about the repository. Carries the persisted consecutive count the
|
|
736
|
+
* ladder decides on (`runtimeUnreachableRung`, docs/reference/specs/resident-repos.md
|
|
737
|
+
* item 64); the message is the named reason, never the SDK's bare
|
|
738
|
+
* `The operation was aborted` — which is what the incident this names sat
|
|
739
|
+
* behind as `degraded(refresh-failed: The operation was aborted)` for forty
|
|
740
|
+
* minutes while nothing escalated. */
|
|
741
|
+
class RuntimeUnreachableError extends Error {
|
|
742
|
+
constructor(
|
|
743
|
+
readonly count: number,
|
|
744
|
+
readonly cause: unknown,
|
|
745
|
+
) {
|
|
746
|
+
super(runtimeUnreachableReason(count));
|
|
747
|
+
this.name = "RuntimeUnreachableError";
|
|
748
|
+
}
|
|
749
|
+
}
|
|
750
|
+
|
|
751
|
+
/** Does this exec rejection mean the control port never answered? The pure
|
|
752
|
+
* signal (`isRuntimeUnreachableSignal`: the `AbortError` name, or the
|
|
753
|
+
* DOMException's message when a wrapper copied only that) over the error and
|
|
754
|
+
* its cause chain. Asked only AFTER `isRuntimeReplacement` — a replaced
|
|
755
|
+
* runtime is a different fact — and only of a spawn-phase error: a
|
|
756
|
+
* command's own output is a `StepError` with an exit code and never gets here. */
|
|
757
|
+
function isRuntimeUnreachable(err: unknown): boolean {
|
|
758
|
+
for (const link of selfAndCauses(err)) if (isRuntimeUnreachableSignal(link)) return true;
|
|
759
|
+
return false;
|
|
760
|
+
}
|
|
717
761
|
/** Trailing slice of one string for an error reason. Command RESULTS are not
|
|
718
762
|
* described here — `describeStepFailure` owns that, because choosing between
|
|
719
763
|
* the two streams is what lost a diagnosis (residentStepReport.ts). */
|
|
@@ -1272,6 +1316,17 @@ const FACTS_KEY = "resident:facts";
|
|
|
1272
1316
|
const SNAPSHOT_KEY = "resident:snapshot";
|
|
1273
1317
|
const DEADLINE_AT_KEY = "resident:provisionDeadlineAt";
|
|
1274
1318
|
const REBUILD_STRIKES_KEY = "resident:rebuildStrikes"; // watchdog auto-rebuild counter
|
|
1319
|
+
/** Consecutive connects the container's control port did not answer (item 64): the ladder's count. */
|
|
1320
|
+
const RUNTIME_UNREACHABLE_KEY = "resident:runtimeUnreachable";
|
|
1321
|
+
|
|
1322
|
+
/** The row under RUNTIME_UNREACHABLE_KEY: the count and both instants, so the
|
|
1323
|
+
* fleet watch sees how long the runtime has been silent. Cleared by any exec
|
|
1324
|
+
* whose process spawned. */
|
|
1325
|
+
interface RuntimeUnreachableRow {
|
|
1326
|
+
count: number;
|
|
1327
|
+
firstAt: string;
|
|
1328
|
+
lastAt: string;
|
|
1329
|
+
}
|
|
1275
1330
|
|
|
1276
1331
|
/** Thread bindings live under their own prefix, keyed by threadKey. */
|
|
1277
1332
|
const THREAD_KEY_PREFIX = "thread:";
|
|
@@ -1713,7 +1768,17 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
1713
1768
|
try {
|
|
1714
1769
|
proc = await createExtensionProcessSandbox(this).exec(argv as unknown as SandboxCommand, launch);
|
|
1715
1770
|
} catch (err) {
|
|
1716
|
-
if (!isRuntimeReplacement(err))
|
|
1771
|
+
if (!isRuntimeReplacement(err)) {
|
|
1772
|
+
// The control port never answered the SDK's connect (its 30 s abort,
|
|
1773
|
+
// raised inside the wake path): no process started and nothing about
|
|
1774
|
+
// the repository is known. Count it in storage — the ladder of item 64
|
|
1775
|
+
// reads the count — and name it, so no reason ever carries the bare
|
|
1776
|
+
// `The operation was aborted`.
|
|
1777
|
+
if (isRuntimeUnreachable(err)) {
|
|
1778
|
+
throw new RuntimeUnreachableError((await this.noteRuntimeUnreachable()).count, err);
|
|
1779
|
+
}
|
|
1780
|
+
throw err;
|
|
1781
|
+
}
|
|
1717
1782
|
// Forward-looking gate, structurally unreachable today: in the pinned SDK
|
|
1718
1783
|
// (@cloudflare/sandbox@0.13.0-next.751.1) every `reason:"runtime_replaced"`
|
|
1719
1784
|
// site hardcodes `retryable:false`, so a replacement currently always
|
|
@@ -1728,6 +1793,9 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
1728
1793
|
);
|
|
1729
1794
|
proc = await createExtensionProcessSandbox(this).exec(argv as unknown as SandboxCommand, launch);
|
|
1730
1795
|
}
|
|
1796
|
+
// The spawn is the proof the control port answers: a persisted count of
|
|
1797
|
+
// unanswered connects ends here, whatever the command goes on to do.
|
|
1798
|
+
await this.clearRuntimeUnreachable();
|
|
1731
1799
|
try {
|
|
1732
1800
|
const out = await proc.output({ encoding: "utf8", timeout: timeout + 30_000 });
|
|
1733
1801
|
// `truncated` is the SDK saying the process log stream was cut past its
|
|
@@ -2404,10 +2472,20 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
2404
2472
|
[DEADLINE_AT_KEY]: systemClock() + provisioningTimeoutMs,
|
|
2405
2473
|
});
|
|
2406
2474
|
await this.ctx.storage.delete([FACTS_KEY, SNAPSHOT_KEY]); // defensive: no stale facts from a past life
|
|
2407
|
-
|
|
2408
|
-
|
|
2409
|
-
|
|
2410
|
-
|
|
2475
|
+
try {
|
|
2476
|
+
this.deleteSchedules(PROVISIONING_CALLBACK);
|
|
2477
|
+
this.deleteSchedules(PROVISION_RUN_CALLBACK);
|
|
2478
|
+
await this.schedule(Math.max(1, Math.ceil(provisioningTimeoutMs / 1000)), PROVISIONING_CALLBACK, resource);
|
|
2479
|
+
await this.schedule(1, PROVISION_RUN_CALLBACK, resource);
|
|
2480
|
+
} catch (err) {
|
|
2481
|
+
// Nothing armed, so nothing may say `onboarding`: the row goes back to
|
|
2482
|
+
// what it was, the way the onboard route frees the registry slot. An
|
|
2483
|
+
// `onboarding` with no schedule behind it used to sit until the
|
|
2484
|
+
// watchdog's provision-timeout (seen live after an offboard in the same
|
|
2485
|
+
// isolate).
|
|
2486
|
+
await this.ctx.storage.delete([RESOURCE_KEY, STATE_KEY, REASON_KEY, UPDATED_KEY, DEADLINE_AT_KEY]);
|
|
2487
|
+
throw err;
|
|
2488
|
+
}
|
|
2411
2489
|
return { state: "onboarding", reason: "" };
|
|
2412
2490
|
}
|
|
2413
2491
|
|
|
@@ -3008,6 +3086,15 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3008
3086
|
* class: `disk-full: …`, never serviceable, and the one failure the
|
|
3009
3087
|
* resident can act on itself (recoverFromDiskFull). */
|
|
3010
3088
|
private async classifyCycleError(err: unknown): Promise<RefreshFailure> {
|
|
3089
|
+
// The control port never answered (item 64): the count decides, and a disk
|
|
3090
|
+
// probe would only cost another 30 s abort against the same silent port.
|
|
3091
|
+
if (err instanceof RuntimeUnreachableError) {
|
|
3092
|
+
return classifyRefreshFailure({
|
|
3093
|
+
step: "refresh",
|
|
3094
|
+
message: err.message,
|
|
3095
|
+
runtimeUnreachable: { count: err.count },
|
|
3096
|
+
});
|
|
3097
|
+
}
|
|
3011
3098
|
return err instanceof StepError
|
|
3012
3099
|
? await this.classifyFailure(err.step, err.message)
|
|
3013
3100
|
: await this.classifyFailure("refresh", errMsg(err));
|
|
@@ -3029,6 +3116,145 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3029
3116
|
if (failure.diskFull) await this.recoverFromDiskFull(failure.reason, selfInFlight);
|
|
3030
3117
|
}
|
|
3031
3118
|
|
|
3119
|
+
// -- the runtime that never answers (docs/reference/specs/resident-repos.md item 64) ------
|
|
3120
|
+
//
|
|
3121
|
+
// Every `sandbox.exec` of the incident this section names rejected after
|
|
3122
|
+
// exactly 30 s with the SDK's connect abort: the WebSocket upgrade to the
|
|
3123
|
+
// container's control port was never answered, for forty minutes, while the
|
|
3124
|
+
// refresh instance recorded `degraded(refresh-failed: The operation was
|
|
3125
|
+
// aborted)` every bucket and nothing escalated — an admin `stop-container`
|
|
3126
|
+
// (a SIGTERM the runtime ignored) did not help either. The recovery is a
|
|
3127
|
+
// ladder over a persisted count of consecutive unanswered connects: re-arm,
|
|
3128
|
+
// stop, destroy and restore from the snapshot, then down with a reason the
|
|
3129
|
+
// watchdog's auto-rebuild strikes apply to. The count lives in storage
|
|
3130
|
+
// because the isolate does not: a Worker deploy or an eviction between
|
|
3131
|
+
// attempts would otherwise restart the ladder at one.
|
|
3132
|
+
|
|
3133
|
+
/** The persisted count of consecutive connects the control port did not
|
|
3134
|
+
* answer, or null while it answers. */
|
|
3135
|
+
private async runtimeUnreachableRow(): Promise<RuntimeUnreachableRow | null> {
|
|
3136
|
+
return (await this.ctx.storage.get<RuntimeUnreachableRow>(RUNTIME_UNREACHABLE_KEY)) ?? null;
|
|
3137
|
+
}
|
|
3138
|
+
|
|
3139
|
+
/** One more unanswered connect: the count up by one, `firstAt` kept, `lastAt`
|
|
3140
|
+
* now — and one log line naming the attempt (the SDK's own line is the bare
|
|
3141
|
+
* AbortError with a stack). */
|
|
3142
|
+
private async noteRuntimeUnreachable(): Promise<RuntimeUnreachableRow> {
|
|
3143
|
+
const prev = await this.runtimeUnreachableRow();
|
|
3144
|
+
const now = new Date(systemClock()).toISOString();
|
|
3145
|
+
const row: RuntimeUnreachableRow = { count: (prev?.count ?? 0) + 1, firstAt: prev?.firstAt ?? now, lastAt: now };
|
|
3146
|
+
await this.ctx.storage.put(RUNTIME_UNREACHABLE_KEY, row);
|
|
3147
|
+
this.runtimeUnreachableSeen = true;
|
|
3148
|
+
console.log(
|
|
3149
|
+
`runtime-unreachable: the control port did not answer within ${SDK_CONNECT_TIMEOUT_MS / 1000} s — attempt ${row.count} of ${RUNTIME_UNREACHABLE_DOWN_AT} (first at ${row.firstAt})`,
|
|
3150
|
+
);
|
|
3151
|
+
return row;
|
|
3152
|
+
}
|
|
3153
|
+
|
|
3154
|
+
/** Whether a row may exist, so the hot path pays one storage read per
|
|
3155
|
+
* isolate and a delete only for a row that is there. Storage stays the
|
|
3156
|
+
* truth; this only says whether it is worth asking. */
|
|
3157
|
+
private runtimeUnreachableSeen: boolean | undefined;
|
|
3158
|
+
|
|
3159
|
+
/** A spawned process is the proof the control port answers: the row goes,
|
|
3160
|
+
* and the log says the silence ended. */
|
|
3161
|
+
private async clearRuntimeUnreachable(): Promise<void> {
|
|
3162
|
+
if (this.runtimeUnreachableSeen === undefined) {
|
|
3163
|
+
this.runtimeUnreachableSeen = (await this.runtimeUnreachableRow()) !== null;
|
|
3164
|
+
}
|
|
3165
|
+
if (!this.runtimeUnreachableSeen) return;
|
|
3166
|
+
const row = await this.runtimeUnreachableRow();
|
|
3167
|
+
await this.ctx.storage.delete(RUNTIME_UNREACHABLE_KEY);
|
|
3168
|
+
this.runtimeUnreachableSeen = false;
|
|
3169
|
+
if (row) {
|
|
3170
|
+
console.log(
|
|
3171
|
+
`runtime-unreachable: cleared — the control port answered again after ${row.count} unanswered attempt(s) since ${row.firstAt}`,
|
|
3172
|
+
);
|
|
3173
|
+
}
|
|
3174
|
+
}
|
|
3175
|
+
|
|
3176
|
+
/** The ladder (`runtimeUnreachableRung`) over the count `run()` persisted,
|
|
3177
|
+
* applied where a refresh step's exec found the port silent. Every rung
|
|
3178
|
+
* records its reason (`lastRefreshError` and the state) and logs one line
|
|
3179
|
+
* naming the rung and the count; the first three then throw the step back
|
|
3180
|
+
* to the engine, whose retry re-enters the same idempotent method thirty
|
|
3181
|
+
* seconds on, doubling — within one instance's six attempts the ladder runs
|
|
3182
|
+
* from the first unanswered connect to `down`, and a count that outlives the
|
|
3183
|
+
* instance carries into the next bucket's. The state is `degraded` under
|
|
3184
|
+
* every rung but the last: a `restoring` marker with no restore running
|
|
3185
|
+
* would hold the cron's instance creation off until the stale bound, so the
|
|
3186
|
+
* retry's wake path flips `restoring` itself when it starts the restore. */
|
|
3187
|
+
private async escalateRuntimeUnreachable(
|
|
3188
|
+
instance: string,
|
|
3189
|
+
step: string,
|
|
3190
|
+
err: RuntimeUnreachableError,
|
|
3191
|
+
): Promise<{ status: "failed"; reason: string }> {
|
|
3192
|
+
const rung = runtimeUnreachableRung(err.count);
|
|
3193
|
+
const reason = runtimeUnreachableReason(err.count, rung);
|
|
3194
|
+
console.log(
|
|
3195
|
+
`refresh instance ${instance}: ${step} runtime-unreachable — rung ${rung} at attempt ${err.count} of ${RUNTIME_UNREACHABLE_DOWN_AT}`,
|
|
3196
|
+
);
|
|
3197
|
+
await this.recordRefreshError(reason);
|
|
3198
|
+
switch (rung) {
|
|
3199
|
+
case "re-arm":
|
|
3200
|
+
await this.setResidentState("degraded", reason);
|
|
3201
|
+
throw err;
|
|
3202
|
+
case "stop":
|
|
3203
|
+
// SIGTERM (`stop()` signals and returns; it cannot kill), the
|
|
3204
|
+
// incarnation swapped: a runtime that still honours signals restarts
|
|
3205
|
+
// under the retry's exec on a fresh disk, and the wake path restores.
|
|
3206
|
+
this.swapIncarnation(); // deliberate incarnation swap
|
|
3207
|
+
await this.stop().catch((stopErr) => console.log(`runtime-unreachable: stop failed: ${errMsg(stopErr)}`));
|
|
3208
|
+
await this.setResidentState("degraded", reason);
|
|
3209
|
+
throw err;
|
|
3210
|
+
case "recreate":
|
|
3211
|
+
await this.recreateContainer(reason);
|
|
3212
|
+
throw err;
|
|
3213
|
+
case "down":
|
|
3214
|
+
// A fresh VM did not answer either. Down with a strike-eligible reason
|
|
3215
|
+
// (REHYDRATION_FAILURE_RE) — the watchdog rebuilds after its passes —
|
|
3216
|
+
// and the VM destroyed, so the rebuild's provisioning starts on a new one.
|
|
3217
|
+
this.swapIncarnation(); // deliberate incarnation swap
|
|
3218
|
+
await this.destroy().catch((destroyErr) =>
|
|
3219
|
+
console.log(`runtime-unreachable: destroy failed: ${errMsg(destroyErr)}`),
|
|
3220
|
+
);
|
|
3221
|
+
await this.clearInstanceLease(instance);
|
|
3222
|
+
return { status: "failed", reason: (await this.goDown(reason)).reason };
|
|
3223
|
+
}
|
|
3224
|
+
}
|
|
3225
|
+
|
|
3226
|
+
/** Destroy the VM and keep everything else: the snapshots, the entry
|
|
3227
|
+
* backups, the registry record, the bindings. `destroy()` is the SDK's
|
|
3228
|
+
* SIGKILL of the whole container (`ctx.container.destroy()`), where `stop()`
|
|
3229
|
+
* is a SIGTERM the runtime may ignore — a control server that no longer
|
|
3230
|
+
* answers its port may not answer signals either, which is what the
|
|
3231
|
+
* incident's admin `stop-container` showed. The disk goes with the VM; the
|
|
3232
|
+
* next exec's wake path finds no runtime, flips `restoring` and restores
|
|
3233
|
+
* mirror, checkout and deps from R2 — the cheap recovery (minutes), where a
|
|
3234
|
+
* rebuild (destroy plus reprovision from the code host) is the expensive
|
|
3235
|
+
* one. The state stays `degraded` with the reason naming the pending
|
|
3236
|
+
* restore, for the reason `escalateRuntimeUnreachable` gives. */
|
|
3237
|
+
private async recreateContainer(reason: string): Promise<void> {
|
|
3238
|
+
console.log(
|
|
3239
|
+
`runtime-unreachable: destroying the container — snapshots kept; the next exec restores from R2 (${reason.slice(0, 200)})`,
|
|
3240
|
+
);
|
|
3241
|
+
this.swapIncarnation(); // deliberate incarnation swap
|
|
3242
|
+
await this.forgetRuntimeIdentity();
|
|
3243
|
+
await this.destroy().catch((err) => console.log(`runtime-unreachable: destroy failed: ${errMsg(err)}`));
|
|
3244
|
+
await this.setResidentState("degraded", reason);
|
|
3245
|
+
}
|
|
3246
|
+
|
|
3247
|
+
/** Forget the SDK's stored runtime identity (`SDK_RUNTIME_RECORD_KEY`)
|
|
3248
|
+
* before a destroy. The SDK's own `stop()` and `destroy()` delete it
|
|
3249
|
+
* (`invalidate`) — this is the guard for the path where they do not get
|
|
3250
|
+
* that far, and it makes the destroy prompt: with no identity stored the
|
|
3251
|
+
* SDK skips the runtime cleanup it would otherwise attempt against the
|
|
3252
|
+
* silent port. Never a recovery on its own: the incident's `stop-container`
|
|
3253
|
+
* had already deleted the record and the next connect aborted the same way. */
|
|
3254
|
+
private async forgetRuntimeIdentity(): Promise<void> {
|
|
3255
|
+
await this.ctx.storage.delete(SDK_RUNTIME_RECORD_KEY);
|
|
3256
|
+
}
|
|
3257
|
+
|
|
3032
3258
|
// -- the refresh cycle as a Workflow instance (item 7) --------------------------
|
|
3033
3259
|
//
|
|
3034
3260
|
// `ResidentRefresh` (the Workflow entrypoint, refresh.ts) calls these
|
|
@@ -3224,6 +3450,20 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3224
3450
|
console.log(`refresh instance ${instance}: ${step} ${err.message}`);
|
|
3225
3451
|
throw err;
|
|
3226
3452
|
}
|
|
3453
|
+
if (err instanceof RuntimeUnreachableError) {
|
|
3454
|
+
// The control port never answered (item 64): the ladder decides — the
|
|
3455
|
+
// step is thrown back for the engine's retry under the first three
|
|
3456
|
+
// rungs, or the resident is down under the last.
|
|
3457
|
+
let result: { status: "failed"; reason: string };
|
|
3458
|
+
try {
|
|
3459
|
+
result = await this.escalateRuntimeUnreachable(instance, step, err);
|
|
3460
|
+
} catch (rethrown) {
|
|
3461
|
+
outcome = `runtime-unreachable (attempt ${err.count} of ${RUNTIME_UNREACHABLE_DOWN_AT}, rung ${runtimeUnreachableRung(err.count)}) — the engine retries`;
|
|
3462
|
+
throw rethrown;
|
|
3463
|
+
}
|
|
3464
|
+
outcome = `failed (${result.reason})`;
|
|
3465
|
+
return { ...result, startedAt, trace: trace.steps() };
|
|
3466
|
+
}
|
|
3227
3467
|
const failure = await this.classifyCycleError(err);
|
|
3228
3468
|
if (failure.interrupted) {
|
|
3229
3469
|
outcome = `interrupted (${failure.reason}) — the engine retries`;
|
|
@@ -3420,6 +3660,10 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3420
3660
|
}): Promise<InstanceStepAnswer<{ result: { evicted: string[]; kept: number } | null; error: string | null }>> {
|
|
3421
3661
|
return this.runInstanceStep(input.instance, "sweep", async (cycle) => {
|
|
3422
3662
|
cycle.count();
|
|
3663
|
+
// A runtime that does not answer has nothing to sweep, and every probe
|
|
3664
|
+
// would cost the SDK's 30 s abort (item 64); the fetch step already
|
|
3665
|
+
// recorded the verdict this instance.
|
|
3666
|
+
if (await this.runtimeUnreachableRow()) return { status: "stopped", why: "runtime-unreachable" };
|
|
3423
3667
|
return this.housekeeping("sweep", () => this.sweepWorktrees(input.resource));
|
|
3424
3668
|
});
|
|
3425
3669
|
}
|
|
@@ -3433,6 +3677,8 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3433
3677
|
}): Promise<InstanceStepAnswer<{ result: { measured: boolean } | null; error: string | null }>> {
|
|
3434
3678
|
return this.runInstanceStep(input.instance, "measure", async (cycle) => {
|
|
3435
3679
|
cycle.count();
|
|
3680
|
+
// Same gate as the sweep: a silent control port cannot answer a `df`.
|
|
3681
|
+
if (await this.runtimeUnreachableRow()) return { status: "stopped", why: "runtime-unreachable" };
|
|
3436
3682
|
return this.housekeeping("measure", async () => ({ measured: (await this.measureDisk()) !== null }));
|
|
3437
3683
|
});
|
|
3438
3684
|
}
|
|
@@ -3942,8 +4188,13 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3942
4188
|
timeoutMs: number,
|
|
3943
4189
|
capBytes?: number,
|
|
3944
4190
|
capFiles?: { out: string; err: string },
|
|
4191
|
+
env?: Record<string, string>,
|
|
3945
4192
|
): Promise<{ stdout: string; stderr: string; exitCode: number; timedOut: boolean; truncated?: boolean }> {
|
|
3946
|
-
|
|
4193
|
+
// The caller's variables (an /exec body's `env`, docs/reference/specs/
|
|
4194
|
+
// harness-pi.md item 4) under the Worker's own: a caller never overrides
|
|
4195
|
+
// what the Worker injects. `su` without `-` keeps this environment for the
|
|
4196
|
+
// thread user's shell.
|
|
4197
|
+
const injected = { ...(env ?? {}), GIT_TERMINAL_PROMPT: "0" };
|
|
3947
4198
|
validateEnvNames(injected);
|
|
3948
4199
|
const body = capBytes
|
|
3949
4200
|
? capWrappedCommand(worktreePath, command, capBytes, capFiles)
|
|
@@ -3965,10 +4216,11 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
3965
4216
|
command: string,
|
|
3966
4217
|
timeoutMs: number,
|
|
3967
4218
|
charCap: number,
|
|
4219
|
+
env?: Record<string, string>,
|
|
3968
4220
|
): Promise<{ stdout: string; stderr: string; exitCode: number; timedOut: boolean; truncated?: boolean }> {
|
|
3969
4221
|
const capBytes = capBytesFor(charCap);
|
|
3970
4222
|
const files = execCapFiles();
|
|
3971
|
-
const r = await this.threadRun(user, worktreePath, command, timeoutMs, capBytes, files);
|
|
4223
|
+
const r = await this.threadRun(user, worktreePath, command, timeoutMs, capBytes, files, env);
|
|
3972
4224
|
if (!r.timedOut) return r;
|
|
3973
4225
|
try {
|
|
3974
4226
|
const rec = await this.threadRun(
|
|
@@ -5130,12 +5382,13 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5130
5382
|
command: string,
|
|
5131
5383
|
timeoutMs: number,
|
|
5132
5384
|
traceparent?: string,
|
|
5385
|
+
env?: Record<string, string>,
|
|
5133
5386
|
): Promise<{ stdout: string; stderr: string; exitCode: number; truncated: boolean } | ThreadErr> {
|
|
5134
5387
|
const queuedAt = systemClock();
|
|
5135
5388
|
let startedAt = queuedAt;
|
|
5136
5389
|
const res = await this.withThreadBusy(threadKey, () => {
|
|
5137
5390
|
startedAt = systemClock();
|
|
5138
|
-
return this.execThreadImpl(threadKey, command, timeoutMs);
|
|
5391
|
+
return this.execThreadImpl(threadKey, command, timeoutMs, env);
|
|
5139
5392
|
});
|
|
5140
5393
|
// The command as the resident's own `resident.exec` root (docs/reference/specs/tracing.md
|
|
5141
5394
|
// item 22): started when the command did, the wait for the thread's turn an attr.
|
|
@@ -5150,6 +5403,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5150
5403
|
threadKey: string,
|
|
5151
5404
|
command: string,
|
|
5152
5405
|
timeoutMs: number,
|
|
5406
|
+
env?: Record<string, string>,
|
|
5153
5407
|
): Promise<{ stdout: string; stderr: string; exitCode: number; truncated: boolean } | ThreadErr> {
|
|
5154
5408
|
const pre = await this.threadPreflight(threadKey);
|
|
5155
5409
|
if ("error" in pre) return pre;
|
|
@@ -5167,7 +5421,7 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
5167
5421
|
|
|
5168
5422
|
let r: Awaited<ReturnType<ResidentDO["threadRun"]>>;
|
|
5169
5423
|
try {
|
|
5170
|
-
r = await this.threadRunCapped(binding.user, binding.worktreePath, command, timeoutMs, EXEC_OUTPUT_CAP);
|
|
5424
|
+
r = await this.threadRunCapped(binding.user, binding.worktreePath, command, timeoutMs, EXEC_OUTPUT_CAP, env);
|
|
5171
5425
|
} catch (err) {
|
|
5172
5426
|
if (err instanceof RuntimeReplacedError) return runtimeReplacedErr(err);
|
|
5173
5427
|
throw err;
|
|
@@ -6011,10 +6265,12 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
6011
6265
|
inFlightKey("hydration"),
|
|
6012
6266
|
LIFECYCLE_KEY,
|
|
6013
6267
|
REFRESH_INSTANCE_KEY,
|
|
6268
|
+
RUNTIME_UNREACHABLE_KEY,
|
|
6014
6269
|
]);
|
|
6015
6270
|
const facts = map.get(FACTS_KEY) as RepoFacts | undefined;
|
|
6016
6271
|
const snap = map.get(SNAPSHOT_KEY) as SnapshotRecord | undefined;
|
|
6017
6272
|
const disk = (map.get(DISK_KEY) as DiskSample | undefined) ?? null;
|
|
6273
|
+
const unreachable = (map.get(RUNTIME_UNREACHABLE_KEY) as RuntimeUnreachableRow | undefined) ?? null;
|
|
6018
6274
|
const refreshRow = (map.get(REFRESH_INSTANCE_KEY) as RefreshInstanceRow | undefined) ?? {
|
|
6019
6275
|
instance: null,
|
|
6020
6276
|
skipped: null,
|
|
@@ -6097,6 +6353,9 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
6097
6353
|
// Item 55: the last disk sample (`residentDiskBudget.ts` DiskSample), or
|
|
6098
6354
|
// null before the first measurement of this incarnation.
|
|
6099
6355
|
disk,
|
|
6356
|
+
// Item 64: consecutive connects the control port did not answer, with
|
|
6357
|
+
// the rung that count is on; null while the port answers.
|
|
6358
|
+
runtimeUnreachable: unreachable ? { ...unreachable, rung: runtimeUnreachableRung(unreachable.count) } : null,
|
|
6100
6359
|
};
|
|
6101
6360
|
}
|
|
6102
6361
|
|
|
@@ -6135,6 +6394,50 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
6135
6394
|
}
|
|
6136
6395
|
}
|
|
6137
6396
|
|
|
6397
|
+
/** Admin `recreate-container` (item 13; item 64's rung 3 on demand): destroy
|
|
6398
|
+
* the VM, keep every snapshot, and start the restore now — the operator's
|
|
6399
|
+
* recovery for a resident whose runtime never answers, minutes where the
|
|
6400
|
+
* rebuild is half an hour. Refused while the engine owns the state
|
|
6401
|
+
* (onboarding/refreshing/restoring — two cycles must never race one disk)
|
|
6402
|
+
* and on a `down` resident, whose one exit is `/rebuild`. The restore runs
|
|
6403
|
+
* in the background through the ordinary wake path (`ensureHydrated`:
|
|
6404
|
+
* `restoring`, mirror, checkout, deps, `warm`, `lastRestore`); the caller
|
|
6405
|
+
* polls `/debug info`. A failure the wake path did not record itself is
|
|
6406
|
+
* recorded here, so the marker never strands `restoring`. */
|
|
6407
|
+
async debugRecreateContainer(): Promise<
|
|
6408
|
+
{ recreated: true; restoreStartedAt: string } | { recreated: false; error: string; status: number }
|
|
6409
|
+
> {
|
|
6410
|
+
const from = await this.getStatus();
|
|
6411
|
+
if (from.state === "onboarding" || from.state === "refreshing" || from.state === "restoring") {
|
|
6412
|
+
return {
|
|
6413
|
+
recreated: false,
|
|
6414
|
+
status: 409,
|
|
6415
|
+
error: `recreate-refused: the engine is mid-flight (state ${from.state}) — retry once it settles (warm/degraded)`,
|
|
6416
|
+
};
|
|
6417
|
+
}
|
|
6418
|
+
if (from.state === "down") {
|
|
6419
|
+
return {
|
|
6420
|
+
recreated: false,
|
|
6421
|
+
status: 409,
|
|
6422
|
+
error: `recreate-refused: the resident is down (${from.reason}) — POST /rebuild is its exit`,
|
|
6423
|
+
};
|
|
6424
|
+
}
|
|
6425
|
+
await this.recreateContainer(
|
|
6426
|
+
"runtime-unreachable: the container was recreated by an operator (recreate-container), snapshots kept — the restore from the snapshot is starting",
|
|
6427
|
+
);
|
|
6428
|
+
const restoreStartedAt = new Date(systemClock()).toISOString();
|
|
6429
|
+
this.ctx.waitUntil(
|
|
6430
|
+
this.ensureHydrated().catch(async (err) => {
|
|
6431
|
+
if (err instanceof ResidentDownError) return; // the wake path recorded it
|
|
6432
|
+
const failure = await this.classifyCycleError(err);
|
|
6433
|
+
console.log(`recreate-container: the restore failed — ${failure.reason.slice(0, 400)}`);
|
|
6434
|
+
await this.recordRefreshError(failure.reason);
|
|
6435
|
+
if ((await this.getStatus()).state === "restoring") await this.setResidentState("degraded", failure.reason);
|
|
6436
|
+
}),
|
|
6437
|
+
);
|
|
6438
|
+
return { recreated: true, restoreStartedAt };
|
|
6439
|
+
}
|
|
6440
|
+
|
|
6138
6441
|
/** Fault injection for the watchdog's auto-rebuild path: persist
|
|
6139
6442
|
* `down` with a rehydration-flavored reason (what a real goDown does), so
|
|
6140
6443
|
* repeated watchdog passes can strike it up to the auto-rebuild without
|
|
@@ -6222,6 +6525,16 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
6222
6525
|
}
|
|
6223
6526
|
}
|
|
6224
6527
|
await this.ctx.storage.delete(REBUILD_STRIKES_KEY);
|
|
6528
|
+
// From scratch means a fresh container too: the SDK's runtime identity
|
|
6529
|
+
// forgotten and the VM destroyed (SIGKILL, a fresh disk) before
|
|
6530
|
+
// provisioning clones onto it. A rebuild that reprovisioned onto the
|
|
6531
|
+
// running container inherited its wedged runtime once — every exec of the
|
|
6532
|
+
// new provisioning met the same unanswered control port (item 64).
|
|
6533
|
+
this.swapIncarnation(); // deliberate incarnation swap
|
|
6534
|
+
await this.forgetRuntimeIdentity();
|
|
6535
|
+
await this.destroy().catch((err) =>
|
|
6536
|
+
console.log(`rebuild: destroy failed (provisioning starts anyway): ${errMsg(err)}`),
|
|
6537
|
+
);
|
|
6225
6538
|
await this.initResident(resource, provisioningTimeoutMs);
|
|
6226
6539
|
return { ...plan, backupObjectsDeleted, state: "onboarding" as const };
|
|
6227
6540
|
}
|
|
@@ -6293,10 +6606,17 @@ export class ResidentDO extends Sandbox<Env> {
|
|
|
6293
6606
|
errors.push(`destroy failed: ${errMsg(err)}`);
|
|
6294
6607
|
}
|
|
6295
6608
|
// Retired DO: clear the alarm the Container base may have armed for its
|
|
6296
|
-
// schedules, then
|
|
6609
|
+
// schedules, then delete every stored key — ours and the SDK's (its runtime
|
|
6610
|
+
// identity among them) — so nothing ever wakes this object again. Keys, not
|
|
6611
|
+
// `deleteAll()`: on a SQLite-backed object that also drops the SDK's
|
|
6612
|
+
// `container_schedules` table, which only its constructor creates, and an
|
|
6613
|
+
// onboard served by this same isolate then fails arming with
|
|
6614
|
+
// `no such table` after writing `onboarding` (seen live). The table stays,
|
|
6615
|
+
// empty — its rows are the two schedules cancelled above.
|
|
6297
6616
|
this.swapIncarnation(); // retired object, retired memos
|
|
6298
6617
|
await this.ctx.storage.deleteAlarm();
|
|
6299
|
-
await this.ctx.storage.
|
|
6618
|
+
const keys = [...(await this.ctx.storage.list()).keys()];
|
|
6619
|
+
for (let i = 0; i < keys.length; i += 128) await this.ctx.storage.delete(keys.slice(i, i + 128));
|
|
6300
6620
|
return { schedulesCancelled: true, containerStopped, storageCleared: true, backupObjectsDeleted, errors };
|
|
6301
6621
|
}
|
|
6302
6622
|
}
|
|
@@ -6735,7 +7055,10 @@ async function handleOnboard(env: Env, body: Record<string, unknown>): Promise<R
|
|
|
6735
7055
|
{
|
|
6736
7056
|
error:
|
|
6737
7057
|
`not-in-installation: the GitHub App cannot mint a token scoped to ${resource.resource} — ` +
|
|
6738
|
-
`
|
|
7058
|
+
`the repository is not in the App installation's repository list, or does not exist under that ` +
|
|
7059
|
+
`exact name (GitHub's token API answers the same 422 for both). An org admin adds it under the ` +
|
|
7060
|
+
`App's installation settings (Settings → GitHub Apps → Configure → Repository access), ` +
|
|
7061
|
+
`then retry (${errMsg(err)})`,
|
|
6739
7062
|
},
|
|
6740
7063
|
403,
|
|
6741
7064
|
);
|
|
@@ -7183,7 +7506,12 @@ async function handleExec(env: Env, body: Record<string, unknown>, traceparent?:
|
|
|
7183
7506
|
// a string) runs at the 5-minute default. A clamp, not a 400: an out-of-range
|
|
7184
7507
|
// ask still runs, at the nearest bound.
|
|
7185
7508
|
const timeoutMs = clampBashTimeout(body.timeoutMs);
|
|
7186
|
-
|
|
7509
|
+
// A caller's extra environment for this one command (docs/reference/specs/
|
|
7510
|
+
// harness-pi.md item 4) — the run bearer the pi harness hands its process —
|
|
7511
|
+
// read from the body alone through the one validated reader the sandbox
|
|
7512
|
+
// Worker uses, and handed to the exec's env option, never onto the command.
|
|
7513
|
+
const execEnv = envFromRequest({ body });
|
|
7514
|
+
return streamThreadExec(ctx.stub.execThread(ctx.threadKey, body.command, timeoutMs, traceparent, execEnv));
|
|
7187
7515
|
}
|
|
7188
7516
|
|
|
7189
7517
|
/** Stream one pending result with the thread-sandbox Worker's heartbeat
|
|
@@ -7423,6 +7751,13 @@ async function handleDebug(env: Env, body: Record<string, unknown>): Promise<Res
|
|
|
7423
7751
|
});
|
|
7424
7752
|
case "stop-container":
|
|
7425
7753
|
return json(await stub.debugStopContainer());
|
|
7754
|
+
case "recreate-container": {
|
|
7755
|
+
// Item 64's rung 3 on demand: destroy the VM, keep the snapshots, start
|
|
7756
|
+
// the restore now. `in` narrowing, as handleRebuild (the RPC stub's
|
|
7757
|
+
// Disposable intersection defeats the boolean discriminant).
|
|
7758
|
+
const r = await stub.debugRecreateContainer();
|
|
7759
|
+
return "error" in r ? json({ error: r.error }, r.status) : json(r, 202);
|
|
7760
|
+
}
|
|
7426
7761
|
case "force-onboarding":
|
|
7427
7762
|
return json(await stub.debugForceOnboarding());
|
|
7428
7763
|
case "force-down": {
|
|
@@ -7468,7 +7803,7 @@ async function handleDebug(env: Env, body: Record<string, unknown>): Promise<Res
|
|
|
7468
7803
|
default:
|
|
7469
7804
|
return json(
|
|
7470
7805
|
{
|
|
7471
|
-
error: `unknown op ${JSON.stringify(op)} (ops: info, schedules, refresh-now, stop-container, force-onboarding, force-down, mint-token, run-watchdog, set-test-overrides, threads, sweep-now, reclaim-now, measure-disk, purge-bindings, backdate-thread, lifecycle)`,
|
|
7806
|
+
error: `unknown op ${JSON.stringify(op)} (ops: info, schedules, refresh-now, stop-container, recreate-container, force-onboarding, force-down, mint-token, run-watchdog, set-test-overrides, threads, sweep-now, reclaim-now, measure-disk, purge-bindings, backdate-thread, lifecycle)`,
|
|
7472
7807
|
},
|
|
7473
7808
|
400,
|
|
7474
7809
|
);
|
|
@@ -30,7 +30,15 @@
|
|
|
30
30
|
// binding's bucket below — the SDK signs URLs for this name, the offboard
|
|
31
31
|
// sweep deletes through the binding. Public config, not secrets.
|
|
32
32
|
"CLOUDFLARE_ACCOUNT_ID": "{{account}}",
|
|
33
|
-
"BACKUP_BUCKET_NAME": "{{script}}-cache"
|
|
33
|
+
"BACKUP_BUCKET_NAME": "{{script}}-cache",
|
|
34
|
+
// The wake budget (docs/reference/specs/resident-repos.md item 64): how long
|
|
35
|
+
// the SDK's physical start waits for the container's control port to accept
|
|
36
|
+
// a request before the first connect — WAKE_PORT_READY_MS in
|
|
37
|
+
// src/execution/residentRefresh.ts, 3 min in place of the SDK's 90 s, so a
|
|
38
|
+
// slow boot (a 2.65 GB image on a cold host) is not read as an unreachable
|
|
39
|
+
// runtime. The env name and its bounds (10 s to 600 s) are the SDK's; a test
|
|
40
|
+
// pins this value to the constant.
|
|
41
|
+
"SANDBOX_PORT_TIMEOUT_MS": "180000"
|
|
34
42
|
},
|
|
35
43
|
"containers": [
|
|
36
44
|
{
|
|
@@ -136,3 +136,21 @@ RUN apt-get update \
|
|
|
136
136
|
&& playwright --version | grep -qx 'Version 1.63.0' \
|
|
137
137
|
&& playwright screenshot --viewport-size=640,480 'data:text/html,<h1>ok</h1>' /tmp/ok.png && test -s /tmp/ok.png \
|
|
138
138
|
&& rm -f /tmp/proof.mp4 /tmp/proof_*.png /tmp/ok.png
|
|
139
|
+
|
|
140
|
+
# pi — the coding harness the bot can start INSIDE this container in place of
|
|
141
|
+
# its native turn loop (docs/reference/specs/harness-pi.md): `pi --mode rpc`,
|
|
142
|
+
# one process per run in the thread's worktree, driven by the bot over its
|
|
143
|
+
# JSONL protocol, with the run's model-proxy bearer as its only key (never a
|
|
144
|
+
# model key: the bearer buys calls through the bot, docs/reference/specs/
|
|
145
|
+
# model-proxy.md). Installed globally with this image's Node — pi needs
|
|
146
|
+
# >= 22.19.0 and the image ships 24 — so every user finds it on PATH, at an
|
|
147
|
+
# EXACT pin held by src/deploy/imagePiHarness.test.ts (the pnpm lesson: a
|
|
148
|
+
# floating tag would move the harness's protocol with the build date, and a
|
|
149
|
+
# pi minor changes RPC events and extension hooks). Proven by the layer, so
|
|
150
|
+
# the BUILD fails, not a run: the pi on PATH answers the pin, and its own help
|
|
151
|
+
# names the RPC mode the harness drives. Dark until a deployment sets
|
|
152
|
+
# `harness: { coding: pi }`: nothing here starts pi on its own.
|
|
153
|
+
RUN npm install -g @earendil-works/pi-coding-agent@0.85.1 \
|
|
154
|
+
&& npm cache clean --force \
|
|
155
|
+
&& pi --version | grep -qx '0.85.1' \
|
|
156
|
+
&& pi --help | grep -q -- '--mode <mode>'
|
|
@@ -75,6 +75,24 @@
|
|
|
75
75
|
"optional": true,
|
|
76
76
|
"note": "The secret half of R2_ACCESS_KEY_ID (same token). Both or neither."
|
|
77
77
|
},
|
|
78
|
+
{
|
|
79
|
+
"name": "ARTIFACTS_R2_ACCESS_KEY_ID",
|
|
80
|
+
"workers": ["bot"],
|
|
81
|
+
"optional": true,
|
|
82
|
+
"note": "R2 API token (S3 access key id) scoped Object Read & Write to the artifacts bucket ONLY (`artifacts.r2.bucket` in config.yaml; docs/reference/specs/execution.md item 20): the bot signs presigned PUT/GET URLs and HEADs objects with it; it never carries a file. Optional: absent with no `artifacts:` section means no store. Created in the Cloudflare dashboard (R2 → Manage API tokens); rotate = new token, `deploy secrets bot`, `deploy restart`."
|
|
83
|
+
},
|
|
84
|
+
{
|
|
85
|
+
"name": "ARTIFACTS_R2_SECRET_ACCESS_KEY",
|
|
86
|
+
"workers": ["bot"],
|
|
87
|
+
"optional": true,
|
|
88
|
+
"note": "The secret half of ARTIFACTS_R2_ACCESS_KEY_ID (same token). Both or neither."
|
|
89
|
+
},
|
|
90
|
+
{
|
|
91
|
+
"name": "ARTIFACTS_COPY_TOKEN",
|
|
92
|
+
"workers": ["bot"],
|
|
93
|
+
"optional": true,
|
|
94
|
+
"note": "Shared bearer between the bot and its own Worker's `POST /artifacts/copy` route, which streams an inbound Slack file into the artifacts bucket (the Worker holds the R2 binding and the Slack token; the bot only asks). Self-minted (`openssl rand -hex 32`); the Worker checks it, the container presents it. Required whenever `artifacts:` is configured — the bot fails fast at startup without it."
|
|
95
|
+
},
|
|
78
96
|
{
|
|
79
97
|
"name": "MCP_CREDENTIAL_KEY",
|
|
80
98
|
"workers": ["bot"],
|