@evident-ai/runner-cdk 3.5.2-dev.8d3aa06 → 3.5.2-dev.bbd799b

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/README.md +91 -40
  2. package/dist/controller-lambda/handler.js +16 -16
  3. package/dist/evident-scale-to-zero-construct.js +7 -0
  4. package/dist/image-version-pruner-lambda/handler.js +28250 -0
  5. package/dist/image-version-reporter-lambda/handler.js +15 -8
  6. package/dist/index.d.ts +1 -1
  7. package/dist/index.js +3 -2
  8. package/dist/microvm/constants.d.ts +1 -1
  9. package/dist/microvm/constants.js +7 -4
  10. package/dist/microvm/construct.d.ts +12 -8
  11. package/dist/microvm/construct.js +42 -33
  12. package/dist/microvm/controller/handle-doorbell.d.ts +2 -2
  13. package/dist/microvm/controller/handle-doorbell.js +2 -2
  14. package/dist/microvm/image/stage-context.d.ts +13 -38
  15. package/dist/microvm/image/stage-context.js +31 -118
  16. package/dist/microvm/image-version-pruner/construct.d.ts +13 -0
  17. package/dist/microvm/image-version-pruner/construct.js +79 -0
  18. package/dist/microvm/image-version-pruner/handler.d.ts +20 -0
  19. package/dist/microvm/image-version-pruner/handler.js +116 -0
  20. package/dist/microvm/image-version-reporter/construct.d.ts +1 -1
  21. package/dist/microvm/image-version-reporter/construct.js +3 -3
  22. package/dist/microvm/image-version-reporter/handler.d.ts +2 -2
  23. package/dist/microvm/image-version-reporter/handler.js +9 -7
  24. package/dist/microvm/shapes.d.ts +10 -21
  25. package/dist/microvm/shapes.js +10 -17
  26. package/dist/waker-lambda/handler.js +3 -3
  27. package/package.json +3 -3
  28. package/dist/microvm-image-context/Dockerfile +0 -190
  29. package/dist/microvm-image-context/hook-server.js +0 -286
  30. package/dist/microvm-image-context/hooks/common.sh +0 -905
  31. package/dist/microvm-image-context/hooks/resume +0 -81
  32. package/dist/microvm-image-context/hooks/run +0 -95
  33. package/dist/microvm-image-context/hooks/suspend +0 -21
  34. package/dist/microvm-image-context/hooks/terminate +0 -36
@@ -1,905 +0,0 @@
1
- #!/usr/bin/env bash
2
- # Sourced by every hook script. Nothing here runs at image build time.
3
- #
4
- # These hooks are the SECOND shell speaking the runner-synchroniser CLI contract
5
- # (runner/docker-images/fargate/entrypoint.sh is the first), so both are held to it by
6
- # runner/synchroniser/src/shell-contract.test.ts — which derives what
7
- # `run_synchroniser` below must handle from the subcommands this shell calls.
8
-
9
- # Installed from npm by the image (docker/Dockerfile's ARG
10
- # RUNNER_SYNCHRONISER_VERSION) and resolved off PATH here. The Dockerfile's
11
- # required-binary assertion is the loud failure if it is missing.
12
- SYNCHRONISER="runner-synchroniser"
13
- OPENCODE_PORT="${OPENCODE_PORT:-4096}"
14
-
15
- # A hook script keeps no memory between invocations, but /resume must re-dial
16
- # with the runner key /run was given, so /run leaves it here. Deliberate, on two
17
- # axes: tmpfs means it never reaches the block device, and the SHARED image
18
- # snapshot is taken at build time — long before /run — so the key cannot enter
19
- # the artefact every VM boots from. And at 0600 owned by uid 10001 it is
20
- # readable by nobody who could not already read it out of the tunnel process's
21
- # /proc/<pid>/environ, which runs as that same uid. /terminate removes it.
22
- # shellcheck disable=SC2034 # read by the scripts that source this file
23
- CONTEXT_FILE="/dev/shm/evident-run-context"
24
-
25
- # The runtime injects MICROVM_ID only into /run, never /resume. This tmpfs file
26
- # survives suspend/resume so /resume can restore the id for every fresh CLI
27
- # process to self-report and acknowledge a fulfilled recycle request (#1906).
28
- # shellcheck disable=SC2034 # read by the scripts that source this file
29
- MICROVM_ID_FILE="/dev/shm/evident-microvm-id"
30
- TUNNEL_PID_FILE="/dev/shm/evident-tunnel.pid"
31
- OPENCODE_PID_FILE="/dev/shm/evident-opencode.pid"
32
- LITESTREAM_PID_FILE="/dev/shm/evident-litestream.pid"
33
-
34
- # Where the runner's credential store lives inside the durable-state bucket.
35
- # The BUCKET is the same for every VM from an image version, so the stack bakes
36
- # it into the image environment; the PREFIX selects one runner's store, so it is
37
- # per-VM and can only arrive in the /run payload. /run leaves it here for the
38
- # hooks that flush back to it. A file of its own rather than a line of
39
- # CONTEXT_FILE: it is not a secret, so /suspend never has to read the runner key
40
- # to find out where to write.
41
- STATE_PREFIX_FILE="/dev/shm/evident-state-prefix"
42
-
43
- # The generated litestream.yml (#812). tmpfs for the same two reasons as the
44
- # files above: it must never reach the block device (Q6, `docs`), and it
45
- # survives suspend/resume, which is what lets `/resume` start litestream again
46
- # with no regeneration cost. The CLI regenerates it only when absent or empty.
47
- LITESTREAM_CONFIG_FILE="/dev/shm/evident-litestream.yml"
48
-
49
- # Set by the CLI when session-DB restore or verification could not prove this
50
- # boot safe to replicate. The teardown hooks read it before flushing, and
51
- # `/terminate` removes it with the rest of the per-VM state.
52
- SESSION_DB_NO_REPLICATE_MARKER="/dev/shm/evident-session-db-no-replicate"
53
-
54
- # Completion evidence for the CLI-owned credential flush. It lives in tmpfs,
55
- # is removed before every handshake, and is also removed by /terminate.
56
- CREDENTIAL_FLUSH_MARKER_FILE="/dev/shm/evident-credential-flush"
57
-
58
- hook_name() { printf '%s' "${0##*/}"; }
59
- log() { echo "[hook:$(hook_name)] $*"; }
60
- warn() { echo "[hook:$(hook_name)] $*" >&2; }
61
- error() { echo "[hook:$(hook_name)] ERROR: $*" >&2; }
62
-
63
- # Returns the CLI's own exit code. `restore` logs domain outcomes and returns 0;
64
- # a non-zero status means the tool itself failed. `sync-once` returns 40 for
65
- # `failed`, `hashFailed`, and `localInvalid` (credentials not persisted), and 0
66
- # for every other outcome. The predicates answer "no" with 10, and
67
- # `session-db-classify`'s three typed answers are 30 (fatal)/31 (replica
68
- # unusable)/32 (retry), extended by `session-db-verify`'s 33 (integrity
69
- # exhausted, replica separated and local disposed) / 34 (could not prove
70
- # separation or disposal). Any status outside a command's contractual answers
71
- # means the tool itself broke, which is the only case worth an ERROR here.
72
- run_synchroniser() {
73
- local rc=0
74
- "${SYNCHRONISER}" "$@" || rc=$?
75
- case "${rc}" in
76
- 0 | 10 | 30 | 31 | 32 | 33 | 34 | 40) ;;
77
- *) error "synchroniser '$*' exited ${rc}; the '${SYNCHRONISER}' command is missing from PATH, corrupt, or it threw" ;;
78
- esac
79
- return "${rc}"
80
- }
81
-
82
- # Wall-clock milliseconds. ${EPOCHREALTIME} is a bash 5 builtin (the image is
83
- # node:22-bookworm-slim → bash 5.2) rather than `date +%s%3N`, which has to
84
- # exec /bin/date inside the caller's command substitution: measured on a loaded
85
- # box, that exec put ~190ms of its own cost INSIDE the window being measured,
86
- # against ~23ms for this — overhead charged to the very number this exists to
87
- # learn. The remaining ~23ms is the command substitution's subshell, kept
88
- # because removing it means an out-parameter global, and ~2% of a ~1.2s call
89
- # biases the reading generous, which is the safe direction for sizing a
90
- # deadline.
91
- #
92
- # The `[.,]` is not paranoia: ${EPOCHREALTIME} renders its decimal separator
93
- # from LC_NUMERIC, so a comma locale would otherwise silently produce garbage
94
- # here. Stripping either turns it into whole microseconds, which /1000 makes
95
- # milliseconds.
96
- #
97
- # Never fails its caller: a timing line is diagnostics, and hardening a
98
- # currently-working path is worse than the gap it closes (#931). The `date`
99
- # fallback covers a pre-5.0 bash; a 0 means "could not read the clock", which
100
- # `log_elapsed_since` turns into `elapsed_ms=unknown` rather than aborting a
101
- # hook under `set -e` — and rather than a 0 that would read as a fast healthy call.
102
- now_ms() {
103
- local now="${EPOCHREALTIME:-}"
104
- if [ -n "${now}" ]; then
105
- now="${now/[.,]/}"
106
- echo "$((now / 1000))"
107
- return 0
108
- fi
109
- date +%s%3N 2>/dev/null || echo 0
110
- }
111
-
112
- # Reports what one pre-warm COST, in a deliberately machine-greppable line
113
- # (`op=`, `elapsed_ms=`, `rc=`) so "how long does this actually take in the
114
- # fleet?" is a log query rather than another spike. `at_s` is where the step
115
- # landed on the hook's own SECONDS clock.
116
- #
117
- # `unknown`, never a number, when either end failed to read the clock (now_ms's
118
- # 0). Subtracting them would report `elapsed_ms=0` — indistinguishable from a
119
- # very fast healthy call, which would quietly bias the fleet-wide sizing data
120
- # this line exists to produce, in the DANGEROUS direction (a deadline sized too
121
- # tight). A non-numeric value drops out of an aggregate instead of poisoning it.
122
- log_elapsed_since() {
123
- local op="$1" started_ms="$2" rc="$3"
124
-
125
- local finished_ms elapsed_ms="unknown"
126
- finished_ms="$(now_ms)"
127
- if [ "${started_ms}" -gt 0 ] && [ "${finished_ms}" -gt 0 ]; then
128
- elapsed_ms="$((finished_ms - started_ms))"
129
- fi
130
-
131
- log "SYNCHRONISER-TIMING op=${op} elapsed_ms=${elapsed_ms} rc=${rc} at_s=${SECONDS}"
132
- }
133
-
134
- # Exports what `runner-synchroniser` resolves its object-store location from
135
- # (runner/synchroniser/src/config.ts). It treats either being empty as
136
- # "persistence disabled" and then reports every restore as a WARNING it still
137
- # exits 0 for — so an unset value has to be caught HERE, where it can still be
138
- # told apart from "the store is simply empty".
139
- load_state_config() {
140
- if [ -z "${LITESTREAM_BUCKET:-}" ]; then
141
- error "LITESTREAM_BUCKET is unset; the image is missing the durable-state bucket the stack bakes in"
142
- return 1
143
- fi
144
-
145
- LITESTREAM_PREFIX="$(cat "${STATE_PREFIX_FILE}" 2>/dev/null || true)"
146
- if [ -z "${LITESTREAM_PREFIX}" ]; then
147
- error "no durable-state prefix at ${STATE_PREFIX_FILE}; /run records the payload's state_prefix there before anything can be restored or flushed"
148
- return 1
149
- fi
150
-
151
- export LITESTREAM_BUCKET LITESTREAM_PREFIX
152
-
153
- # Mirrors runner-synchroniser's own persistenceEnabled predicate (config.ts):
154
- # both non-empty, already checked above, so by this point PERSISTENCE_BUCKET
155
- # is just LITESTREAM_BUCKET. The CLI's replicator and flush_session_db (#812
156
- # WI-3/WI-4) read ONLY this var, never LITESTREAM_BUCKET/LITESTREAM_PREFIX
157
- # directly, so every entry point agrees on one predicate regardless of which
158
- # earlier step in THIS hook process set it; /resume, /suspend and /terminate
159
- # all use this same function before any teardown decision.
160
- export PERSISTENCE_BUCKET="${LITESTREAM_BUCKET}"
161
- }
162
-
163
- # Best-effort by design: the teardown hooks must still complete even when a
164
- # credential flush cannot happen.
165
- sync_credentials() {
166
- load_state_config || return 0
167
- run_synchroniser sync-once claude || true
168
- run_synchroniser sync-once opencode || true
169
- }
170
-
171
- # --- boot pre-warms ----------------------------------------------------------
172
-
173
- # Reads the ~30 MB litestream binary into the page cache, in the background, so
174
- # the FIRST exec of it does not pay that read on the critical path.
175
- #
176
- # The pre-warm overlaps the CLI's node startup and auth round trip, so the first
177
- # litestream operation does not also pay the binary's cold page-cache read.
178
- # Boot measurements put ~6.6-7.3s before the litestream version line, of which
179
- # only ~0.2s is accounted for by the two node calls in between — the remainder
180
- # is INFERRED to be this read, never measured.
181
- # The `litestream-prewarm` and `litestream-version` timings are what settle it
182
- # on the next boot.
183
- #
184
- # Nothing waits on it and nothing reads its output, so if that inference is
185
- # wrong this costs one backgrounded `cat`. Deleting the `litestream version`
186
- # diagnostic instead would save nothing: `litestream restore` two calls later
187
- # pays the identical read.
188
- prewarm_litestream() {
189
- local binary
190
- binary="$(command -v litestream 2>/dev/null || true)"
191
- if [ -z "${binary}" ]; then
192
- warn "litestream is not on PATH; skipping the boot pre-warm"
193
- return 0
194
- fi
195
-
196
- # `cat` rather than a throwaway `litestream version`: a sequential read gets
197
- # the whole file with readahead, where an exec demand-pages it.
198
- (
199
- local started_ms rc=0
200
- started_ms="$(now_ms)"
201
- cat "${binary}" >/dev/null 2>&1 || rc=$?
202
- log_elapsed_since litestream-prewarm "${started_ms}" "${rc}"
203
- ) &
204
- log "pre-warming ${binary} in the background"
205
- }
206
-
207
- # Reads the aws CLI v2 install tree into the page cache, in the background, so
208
- # the CLI's first runner-secret fetch does not pay first-touch I/O. The read
209
- # overlaps `evident run`'s node startup and auth round trip rather than a hook
210
- # deadline.
211
- #
212
- # A single `cat` of the `aws` entrypoint (litestream's pattern above) is NOT
213
- # enough here: v2 ships as a real Python distribution (~7,500 files under the
214
- # resolved binary's own directory), and a cold invocation demand-pages a
215
- # scattered set of them (botocore's endpoints.json/partitions.json, service
216
- # model JSON, shared libs) — one boot measured #1997's fetch at 7,243ms cold
217
- # vs ~400ms warm, an ~18x gap the litestream read-ahead trick alone cannot
218
- # close. `find -exec cat` walks that whole tree instead of one file.
219
- prewarm_aws_cli() {
220
- local binary tree
221
- binary="$(command -v aws 2>/dev/null || true)"
222
- if [ -z "${binary}" ]; then
223
- warn "aws CLI is not on PATH; skipping the boot pre-warm"
224
- return 0
225
- fi
226
- tree="$(dirname "$(readlink -f "${binary}")")"
227
-
228
- (
229
- local started_ms rc=0
230
- started_ms="$(now_ms)"
231
- find "${tree}" -type f -exec cat {} + >/dev/null 2>&1 || rc=$?
232
- log_elapsed_since aws-cli-prewarm "${started_ms}" "${rc}"
233
- ) &
234
- log "pre-warming ${tree} in the background"
235
- }
236
-
237
- # `kill -0` answers "does this pid exist", which is not the question any caller
238
- # here is asking. A process that has exited but has not been reaped — a zombie —
239
- # still exists, so `kill -0` reports a corpse as ALIVE. That condition is the
240
- # normal case for everything these hooks start: `start_tunnel` and its CLI child
241
- # background a process that outlives the hook, the hook shell must return so AWS
242
- # gets its 200, and PID 1 in this image is a bare node hook server with no init
243
- # (Dockerfile) — so nothing ever wait()s for the orphan. It stayed a zombie for
244
- # the rest of the VM's life, which made `stop_tunnel` burn its whole budget
245
- # SIGKILLing a corpse and made a stale pid file suppress the next
246
- # `start_tunnel`, leaving a resumed VM permanently offline (#718).
247
- #
248
- # This is zombie-safe, NOT identity-safe: a recycled pid reads R/S and is still
249
- # called alive, so the SIGKILL backstop could in principle hit an unrelated
250
- # process. Pre-existing limitation of the pid-file approach, out of scope here —
251
- # named so nobody over-trusts the helper (src/image/hook-scripts.test.ts).
252
- #
253
- # Also sets PROC_DEAD_REASON on every return path — "gone" for a `kill -0`
254
- # miss, "zombie" for an unreaped `Z`, reset to empty on the alive path so a
255
- # stale value from a previous call can never be read — for the caller that
256
- # needs to say WHICH death it was (stop_tunnel's outcome line), never for this
257
- # function itself: see the silence note below. File-scope init (below) is what
258
- # keeps a read of it safe under `set -u` before this has ever run.
259
- PROC_DEAD_REASON=""
260
-
261
- process_is_alive() {
262
- if ! kill -0 "$1" 2>/dev/null; then
263
- PROC_DEAD_REASON="gone"
264
- return 1
265
- fi
266
-
267
- # `/proc/<pid>/status` rather than `/proc/<pid>/stat`, whose fields cannot be
268
- # split safely when a comm contains a space or a paren. If the state cannot be
269
- # read at all, keep the `kill -0` answer: absent evidence is not "dead", and
270
- # reading it as dead would SIGTERM-and-forget a live, still-draining CLI.
271
- # Three fields, not two: `read` hands the whole remainder to its LAST variable,
272
- # so a trailing `_` is what keeps the state letter out of `Z (zombie)`.
273
- #
274
- # That deferral is silent HERE on purpose: `stop_tunnel` polls this 10x/s, so a
275
- # warning inside it would bury the log. The flag lets `is_running` say it once
276
- # per guard instead — absent evidence must warn as well as defer
277
- # (development-workflow.mdc).
278
- PROC_STATE_UNREADABLE=0
279
- local state
280
- read -r _ state _ < <(grep -m1 '^State:' "/proc/$1/status" 2>/dev/null) || {
281
- PROC_STATE_UNREADABLE=1
282
- PROC_DEAD_REASON="" # unreadable /proc still means "alive" below, not a dead reason
283
- return 0
284
- }
285
-
286
- if [ "${state}" = "Z" ]; then
287
- PROC_DEAD_REASON="zombie"
288
- return 1
289
- fi
290
-
291
- PROC_DEAD_REASON=""
292
- return 0
293
- }
294
-
295
- is_running() {
296
- [ -s "$1" ] || return 1
297
-
298
- local pid
299
- pid="$(cat "$1")"
300
- process_is_alive "${pid}" || return 1
301
-
302
- # The one state in which this whole helper is back to being a bare `kill -0`
303
- # (a `hidepid=` mount, a `/proc` that is not mounted): every liveness answer
304
- # silently reverts to #718 — a corpse reads as alive, so suspend SIGKILLs it
305
- # after the full wait and the next start is suppressed for the VM's life.
306
- [ "${PROC_STATE_UNREADABLE}" = 0 ] \
307
- || warn "could not read /proc/${pid}/status; falling back to 'kill -0', which reports an exited-but-unreaped process as alive (#718)"
308
-
309
- return 0
310
- }
311
-
312
- # The snapshot bakes one machine id into every VM launched from this image
313
- # version, so /run replaces it. Best-effort per file: the hook runs as uid 10001
314
- # and these live in root-owned directories, so a VM that kept the snapshot's id
315
- # must still boot — but it says so, because it is a state worth being able to see.
316
- regenerate_machine_id() {
317
- local machine_id
318
- machine_id="$(tr -d '-' </proc/sys/kernel/random/uuid)"
319
-
320
- for path in /etc/machine-id /var/lib/dbus/machine-id; do
321
- printf '%s\n' "${machine_id}" >"${path}" 2>/dev/null \
322
- || warn "could not rewrite ${path}; this VM keeps the snapshot's machine id"
323
- done
324
- }
325
-
326
- tunnel_is_running() { is_running "${TUNNEL_PID_FILE}"; }
327
- opencode_is_running() { is_running "${OPENCODE_PID_FILE}"; }
328
- litestream_is_running() { is_running "${LITESTREAM_PID_FILE}"; }
329
-
330
- # `jq -e` alone is not enough: its exit status reflects the LAST OUTPUT VALUE,
331
- # and an interpolation of a missing field is still a non-empty string, so a
332
- # payload with no runner_key would sail through as the literal "null".
333
- payload_is_complete() {
334
- jq -e '
335
- (.runner_key | type == "string" and length > 0)
336
- and (.endpoints.api | type == "string" and length > 0)
337
- and (.endpoints.tunnel | type == "string" and length > 0)
338
- and (.state_prefix | type == "string" and length > 0)
339
- ' >/dev/null 2>&1
340
- }
341
-
342
- stop_opencode() {
343
- if ! opencode_is_running; then
344
- # Clear the file here too, for the same reason stop_tunnel does: opencode
345
- # has exit paths that never reach this function (a crash, an OOM kill), so
346
- # the pid file routinely outlives the process it names, and leaving it is
347
- # #718 again the moment anything starts reaping orphans and the pid is
348
- # recycled.
349
- rm -f "${OPENCODE_PID_FILE}"
350
- log "no opencode to stop"
351
- return 0
352
- fi
353
-
354
- kill -TERM "$(cat "${OPENCODE_PID_FILE}")" 2>/dev/null || true
355
- rm -f "${OPENCODE_PID_FILE}"
356
- log "opencode stopped"
357
- }
358
-
359
- # The ceiling stop_opencode_and_wait (below) waits before its SIGKILL backstop.
360
- # /terminate can afford a real drain — the VM is being torn down either way —
361
- # unlike /run's cleanup, which cannot (stop_opencode's own cheapness above is
362
- # load-bearing for that budget, §4, and must not grow a wait).
363
- OPENCODE_STOP_WAIT_SECONDS="${EVIDENT_OPENCODE_STOP_WAIT_SECONDS:-5}"
364
-
365
- # The graceful stop /terminate needs (#812 WI-4): SIGTERM, poll for exit,
366
- # SIGKILL backstop, clear the pid file. A NEW function rather than growing
367
- # stop_opencode itself, for the reason above — mirrors stop_tunnel's and
368
- # stop_litestream's shape, including PROC_DEAD_REASON's one deciding read
369
- # after the poll loop (never inside it, which would bury the log at 10x/s).
370
- # This is half of the ordered-shutdown invariant (plan §6): /terminate must
371
- # stop opencode BEFORE flush_session_db's final sync, or litestream could
372
- # snapshot while opencode is still writing its WAL and the last session
373
- # writes would be missing from S3.
374
- stop_opencode_and_wait() {
375
- if ! opencode_is_running; then
376
- rm -f "${OPENCODE_PID_FILE}"
377
- log "no opencode to stop"
378
- return 0
379
- fi
380
-
381
- local pid
382
- pid="$(cat "${OPENCODE_PID_FILE}")"
383
- kill -TERM "${pid}" 2>/dev/null || true
384
-
385
- local tenths=0
386
- local deadline_tenths=$(( OPENCODE_STOP_WAIT_SECONDS * 10 ))
387
- while [ "${tenths}" -lt "${deadline_tenths}" ]; do
388
- process_is_alive "${pid}" || break
389
- sleep 0.1
390
- tenths=$(( tenths + 1 ))
391
- done
392
-
393
- local outcome
394
- if process_is_alive "${pid}"; then
395
- warn "opencode ${pid} ignored SIGTERM; killing"
396
- kill -KILL "${pid}" 2>/dev/null || true
397
- outcome="after SIGKILL (ignored SIGTERM for ${OPENCODE_STOP_WAIT_SECONDS}s)"
398
- else
399
- outcome="cleanly in $(( tenths / 10 )).$(( tenths % 10 ))s (${PROC_DEAD_REASON})"
400
- fi
401
-
402
- rm -f "${OPENCODE_PID_FILE}"
403
- log "opencode stopped ${outcome}"
404
- }
405
-
406
- # The graceful wait before the SIGKILL backstop in stop_litestream, below.
407
- # litestream's own sync is normally sub-second, so 10s is generous headroom —
408
- # chosen the same way TUNNEL_STOP_WAIT_SECONDS was, not a measured worst case.
409
- LITESTREAM_STOP_WAIT_SECONDS="${EVIDENT_LITESTREAM_STOP_WAIT_SECONDS:-10}"
410
-
411
- # The graceful stop: mirrors stop_tunnel exactly, including PROC_DEAD_REASON's
412
- # one deciding read AFTER the poll loop (never inside it, which polls 10x/s and
413
- # would bury the log) — gone / zombie / SIGKILLed are the three outcomes an
414
- # operator needs told apart, not a flatter clean/killed binary. /suspend and
415
- # /terminate call this: the SIGTERM it sends IS the checked flush they need.
416
- stop_litestream() {
417
- if ! litestream_is_running; then
418
- rm -f "${LITESTREAM_PID_FILE}"
419
- log "no litestream to stop"
420
- return 0
421
- fi
422
-
423
- local pid
424
- pid="$(cat "${LITESTREAM_PID_FILE}")"
425
- kill -TERM "${pid}" 2>/dev/null || true
426
-
427
- local tenths=0
428
- local deadline_tenths=$(( LITESTREAM_STOP_WAIT_SECONDS * 10 ))
429
- while [ "${tenths}" -lt "${deadline_tenths}" ]; do
430
- process_is_alive "${pid}" || break
431
- sleep 0.1
432
- tenths=$(( tenths + 1 ))
433
- done
434
-
435
- local outcome
436
- if process_is_alive "${pid}"; then
437
- warn "litestream ${pid} ignored SIGTERM; killing"
438
- kill -KILL "${pid}" 2>/dev/null || true
439
- outcome="after SIGKILL (ignored SIGTERM for ${LITESTREAM_STOP_WAIT_SECONDS}s)"
440
- else
441
- outcome="cleanly in $(( tenths / 10 )).$(( tenths % 10 ))s (${PROC_DEAD_REASON})"
442
- fi
443
-
444
- rm -f "${LITESTREAM_PID_FILE}"
445
- log "litestream stopped ${outcome}"
446
- }
447
-
448
- # The cheap stop: one SIGTERM, one rm, one log line, NO poll loop — `/run`'s
449
- # own cleanup calls this, never stop_litestream, because that path only runs
450
- # after /run has already FAILED: no session happened yet, so a graceful final
451
- # sync is worth nothing against a SIGTERM budget that cannot afford another
452
- # wait on top of stop_tunnel's own. A distinct function rather than a mode
453
- # argument on stop_litestream:
454
- # stop_tunnel's own comment argues against an argument a future caller can get
455
- # wrong. litestream_is_running is the same O(1) check stop_litestream's own
456
- # no-op branch uses, not a poll — /run's cleanup trap fires this before
457
- # the replicator start is ever reached whenever an earlier step failed, and an
458
- # operator reading that log must not be told a stop signal went to a process
459
- # that never started.
460
- kill_litestream() {
461
- if ! litestream_is_running; then
462
- rm -f "${LITESTREAM_PID_FILE}"
463
- log "no litestream to stop"
464
- return 0
465
- fi
466
-
467
- kill -TERM "$(cat "${LITESTREAM_PID_FILE}")" 2>/dev/null || true
468
- rm -f "${LITESTREAM_PID_FILE}"
469
- log "litestream stop signalled (no wait)"
470
- }
471
- # --- litestream replicate (end) ----------------------------------------------
472
-
473
- # --- flush_session_db (#812 WI-4) -------------------------------------------
474
- #
475
- # The checked, synchronous flush /suspend and /terminate need before they
476
- # finish tearing down. Never fatal, for the same reason every session-DB
477
- # function in this file is: /suspend and /terminate must complete their own
478
- # teardown regardless of whether S3 could be reached.
479
-
480
- # Ceiling for the one-shot `litestream replicate -once` flush below. litestream's
481
- # own sync is normally sub-second (M5, the plan's grounding); 10s is generous
482
- # headroom, chosen the same way LITESTREAM_STOP_WAIT_SECONDS was above — not a
483
- # measured worst case. The SIGKILL backstop gives the flush its own two-second
484
- # grace after `timeout` sends SIGTERM.
485
- SESSION_DB_FLUSH_DEADLINE_SECONDS="${EVIDENT_SESSION_DB_FLUSH_DEADLINE_SECONDS:-10}"
486
- SESSION_DB_FLUSH_KILL_GRACE_SECONDS=2
487
-
488
- # Guards mirror the CLI replicator's first three exactly (disabled -> marker ->
489
- # config) — a flush must never attempt work those guards would have refused
490
- # to start in the first place. PERSISTENCE_BUCKET is the SAME variable
491
- # the CLI reads, so both agree regardless of which earlier step in
492
- # THIS process set it (load_state_config above).
493
- #
494
- # Then: stop the daemon FIRST — its SIGTERM IS litestream's own final-sync
495
- # trigger — and only once it is confirmed stopped does the synchronous -once
496
- # flush run, as a second, checked pass. Running -once while the daemon is
497
- # still alive would put two writers on one prefix, the single-writer
498
- # invariant this whole feature exists to protect (Q8) — this ordering is not
499
- # negotiable.
500
- #
501
- # If the daemon has already died (a missing or stale pid file), that IS the
502
- # detection this needs: SESSION-DB-REPLICATOR-DIED names it loudly instead of
503
- # silently skipping straight to the flush (development-workflow.mdc: every
504
- # recovery branch emits a server-visible signal). The flush still runs either
505
- # way — it is what actually gets any unreplicated writes to S3 before the
506
- # caller's next step.
507
- flush_session_db() {
508
- if [ -z "${PERSISTENCE_BUCKET:-}" ]; then
509
- warn "SESSION-DB-PERSISTENCE-DISABLED: LITESTREAM_BUCKET/LITESTREAM_PREFIX are not both set; nothing to flush."
510
- return 0
511
- fi
512
-
513
- if [ -e "${SESSION_DB_NO_REPLICATE_MARKER}" ]; then
514
- log "skipping session-DB flush: this boot's session DB was not proven safe to replicate (see the SESSION-DB-* warning above)"
515
- return 0
516
- fi
517
-
518
- if [ ! -s "${LITESTREAM_CONFIG_FILE}" ]; then
519
- error "no usable ${LITESTREAM_CONFIG_FILE}; cannot flush the session DB"
520
- return 0
521
- fi
522
-
523
- if litestream_is_running; then
524
- stop_litestream
525
- else
526
- warn "SESSION-DB-REPLICATOR-DIED: no live litestream at ${LITESTREAM_PID_FILE}; flushing one-shot anyway"
527
- fi
528
-
529
- # Only reached with the daemon confirmed stopped, or never running — never
530
- # concurrently with it (see above).
531
- local flush_rc=0
532
- timeout -k "${SESSION_DB_FLUSH_KILL_GRACE_SECONDS}" "${SESSION_DB_FLUSH_DEADLINE_SECONDS}" \
533
- litestream replicate -once -config "${LITESTREAM_CONFIG_FILE}" || flush_rc=$?
534
-
535
- if [ "${flush_rc}" -eq 0 ]; then
536
- log "SESSION-DB-FINAL-FLUSH: succeeded, ${SECONDS}s into the hook"
537
- else
538
- warn "SESSION-DB-FINAL-FLUSH: litestream replicate -once exited ${flush_rc}, ${SECONDS}s into the hook; some writes since the last sync may not have reached S3"
539
- fi
540
- }
541
- # --- flush_session_db (end) --------------------------------------------------
542
-
543
- # How long the guest CLI runs with no activity before it exits itself
544
- # (`evident run --idle-timeout`), which is what
545
- # turns a truly-abandoned VM into the clean-offline POST that lets Evident
546
- # suspend it (#732). Sized from the measured cost of guessing wrong rather than
547
- # the saving: suspend reaches SUSPENDED in ~7 s and a resume is RUNNING in
548
- # ~0.6 s with the tunnel back ~2 s later (README, "Measured, end to end"), so
549
- # napping a VM whose user comes straight back costs ~10 s — against the
550
- # legacy conservative baseline input of ~$0.30/h, AWS can burst a loaded VM
551
- # to a 16 GB / 8 vCPU peak (~$1.06/h at sustained full load—a worst-case
552
- # ceiling, not an expectation); that is cheap enough that the balance
553
- # sits far nearer the floor than the ceiling. Not AT the floor, though: the
554
- # CLI's idle detector needs 2 clear poll cycles (≥4 s of real time), so a value
555
- # near that would spend more time
556
- # suspending/resuming than idle.
557
- # ECS's waker uses 900 s instead only because *its* cold start is far slower
558
- # than this VM's ~2 s resume — not evidence this default should match it.
559
- #
560
- # NOT part of the hook's own teardown budget (TUNNEL_STOP_WAIT_SECONDS et al.,
561
- # below): it governs the CLI's lifetime long after the hook has already
562
- # returned its 200, so it is deliberately outside that arithmetic.
563
- #
564
- # EVIDENT_IDLE_TIMEOUT_SECONDS, deliberately the SAME env var name
565
- # runner/docker-images/fargate/entrypoint.sh uses for the ECS runner's own
566
- # idle-timeout flag, so an operator who knows one knows the other. A
567
- # non-numeric override must never silently DROP the flag — that degrades to
568
- # an always-on VM burning ~$7.17/day at baseline, or up to ~$25.49/day at
569
- # sustained full peak as load increases (a worst-case ceiling, not an
570
- # expectation), exactly the bug this closes — so it
571
- # warns and falls back to the default instead.
572
- #
573
- # Deliberately NOT in the /run payload yet: doing so would touch the doorbell
574
- # contract in four coordinated places for no capability today. #716 shipped
575
- # the swarm/per-user provisioner but explicitly scoped control-plane
576
- # idle/suspend policy OUT (its own "Phase 3" — see the issue's "Out of
577
- # Scope"), so a per-pool idle policy is still a clean follow-up, not
578
- # something #716 already provides a home for.
579
- IDLE_TIMEOUT_SECONDS=120
580
- if [ -n "${EVIDENT_IDLE_TIMEOUT_SECONDS:-}" ]; then
581
- if [[ "${EVIDENT_IDLE_TIMEOUT_SECONDS}" =~ ^[0-9]+$ ]]; then
582
- IDLE_TIMEOUT_SECONDS="${EVIDENT_IDLE_TIMEOUT_SECONDS}"
583
- else
584
- warn "EVIDENT_IDLE_TIMEOUT_SECONDS='${EVIDENT_IDLE_TIMEOUT_SECONDS}' is not numeric; using the default ${IDLE_TIMEOUT_SECONDS}s instead of leaving the VM always-on"
585
- fi
586
- fi
587
-
588
- # `setsid` so the tunnel outlives this hook: the script must return so AWS gets
589
- # its 200, while the tunnel keeps serving.
590
- start_tunnel() {
591
- local runner_key="$1" api_url="$2" tunnel_url="$3"
592
- local restore_runner_credentials="${4:-false}" restore_history="${5:-false}"
593
- local -a credential_flags=()
594
- local -a session_db_flags=()
595
- local -a opencode_config_flags=()
596
-
597
- # The explicit fourth argument is set only by /run; /resume uses the default
598
- # so a resumed VM never restores credentials over stores it already has.
599
- if [ "${restore_runner_credentials}" = true ]; then
600
- credential_flags=(--restore-runner-credentials)
601
- fi
602
-
603
- # Only a fresh /run asks the CLI to restore session history. /resume keeps
604
- # the database from the snapshot and must not restore over it.
605
- if [ "${restore_history}" = true ]; then
606
- session_db_flags=(--restore-session-db)
607
- fi
608
-
609
- if [ -n "${RUNNER_OPENCODE_CONFIG:-}" ]; then
610
- opencode_config_flags=(--opencode-config-overlay "${RUNNER_OPENCODE_CONFIG}")
611
- fi
612
-
613
- if tunnel_is_running; then
614
- warn "tunnel already running (pid $(cat "${TUNNEL_PID_FILE}")); not starting a second one"
615
- return 0
616
- fi
617
-
618
- # The key travels in the child's environment ONLY: argv is world-readable
619
- # through /proc.
620
- # --enable-file-sync-to is what makes the UI's "connect a provider account"
621
- # flow reach this VM: without it the CLI declines every queued file and the
622
- # page can only say "this runner isn't accepting files". There is no shell
623
- # into a MicroVM, so it is the ONLY way to seed model credentials on a runner
624
- # that is already up — the S3 credential store is read at /run and /resume,
625
- # never mid-life. The allow-list stays a single directory, as on ECS
626
- # (runner/docker-images/fargate/entrypoint.sh): the flag is repeatable, but every
627
- # extra entry widens what Evident can write into a VM that executes agent
628
- # code. ${HOME} is set by the image (ENV HOME=/home/runner) and `set -u` makes
629
- # an unset one abort rather than silently allow-list "/.claude".
630
- EVIDENT_RUNNER_KEY="${runner_key}" setsid evident run \
631
- --port "${OPENCODE_PORT}" \
632
- --endpoint "${api_url}" \
633
- --tunnel "${tunnel_url}" \
634
- --opencode-pid-file "${OPENCODE_PID_FILE}" \
635
- --litestream-config "${LITESTREAM_CONFIG_FILE}" \
636
- --litestream-pid-file "${LITESTREAM_PID_FILE}" \
637
- --session-db-no-replicate-marker "${SESSION_DB_NO_REPLICATE_MARKER}" \
638
- --credential-sync-marker "${CREDENTIAL_FLUSH_MARKER_FILE}" \
639
- "${credential_flags[@]}" \
640
- "${session_db_flags[@]}" \
641
- "${opencode_config_flags[@]}" \
642
- --idle-timeout "${IDLE_TIMEOUT_SECONDS}" \
643
- --enable-file-sync-to "${HOME}/.claude" &
644
-
645
- echo $! >"${TUNNEL_PID_FILE}"
646
- log "tunnel started (pid $(cat "${TUNNEL_PID_FILE}"))"
647
- }
648
-
649
- # The worst case `evident run` can take to shut down gracefully on SIGTERM, in
650
- # whole seconds: 25 s in-flight drain (SHUTDOWN_DRAIN_TIMEOUT_MS) + 2 s offline
651
- # POST (notifyAgentDisconnected) + 5 s telemetry flush
652
- # (TELEMETRY_SHUTDOWN_TIMEOUT_MS) + 8.5 s pre-drain credential flush
653
- # + 8.5 s post-drain credential flush. Each phase is bounded there, so this is
654
- # a ceiling rather than a typical cost — an idle suspend finishes in a couple
655
- # of seconds. This is the ONE place the hand-maintained budget is written down;
656
- # a guard in the source repository keeps this number in step with the CLI's
657
- # declared bounds.
658
- #
659
- # DOCUMENTATION ONLY — nothing is derived from this value. It is the hand-maintained
660
- # sum of the five bounded phases above (25 + 2 + 5 + 8.5 + 8.5); update it here if
661
- # any of those bounds changes.
662
- # shellcheck disable=SC2034 # documentation; deliberately read by nothing
663
- CLI_SHUTDOWN_CEILING_SECONDS=49
664
-
665
- # How long stop_tunnel waits for that shutdown before the SIGKILL backstop.
666
- # Since #718 this binds ONLY for a CLI that is still draining: one that has
667
- # already exited is detected on the first 0.1 s poll whatever this says, because
668
- # process_is_alive no longer mistakes its unreaped corpse for a live process.
669
- # That is what takes a routine suspend from ~41 s back to the ~2 s it measures.
670
- #
671
- # So 10 is chosen, not derived, and it sits in the middle of a real range. The
672
- # measured shutdown was 1255 ms of drain inside a 2322 ms total (#697): 5 s is
673
- # only ~2.2x that and re-arms #657's complaint of a kill landing mid-drain,
674
- # while 30 s never truncates a legitimate drain but rebuilds the headroom
675
- # problem #699 created — 41 s against a 60 s lifecycleTimeout leaves 19 s for
676
- # everything else in the hook. 10 s is ~4x the measured shutdown and gives a
677
- # provable worst case of 10 + the 10 s `suspend` spends in sync_credentials
678
- # before us = 20 s, inside the ~55 s the in-VM server allows this script
679
- # (DEFAULT_TIMEOUT_SECONDS, aws/lambda-microvm-runtime/src/runtime.ts),
680
- # itself inside AWS's 60 s (HOOK_TIMEOUT_SECONDS, src/constants.ts) —
681
- # overrunning that fails the whole lifecycle transition, which is worse than the
682
- # kill this replaces. The residual cost is a drain longer than 10 s being
683
- # truncated; that work stays `processing` server-side and is re-adopted on the
684
- # next start (ADR-0046).
685
- #
686
- # src/image/hook-scripts.test.ts holds both that this stays small and that it
687
- # fits the hook timeout, and times a real SIGKILL against the override to prove
688
- # the loop HONOURS it rather than a hardcoded deadline. The env override is for
689
- # tests only, so they need not burn 10 s of wall clock.
690
- TUNNEL_STOP_WAIT_SECONDS="${EVIDENT_TUNNEL_STOP_WAIT_SECONDS:-10}"
691
-
692
- stop_tunnel() {
693
- if ! tunnel_is_running; then
694
- # Clear the file here too, not only on the path below: `evident run` has
695
- # self-exit paths that never reach this function (auth expired, idle
696
- # timeout, a crash), so the pid file routinely outlives the process it
697
- # names. Leaving it means every later guard re-reads a dead pid — harmless
698
- # while process_is_alive agrees it is dead, and #718 again the moment
699
- # anything starts reaping orphans and the pid gets recycled.
700
- rm -f "${TUNNEL_PID_FILE}"
701
- log "no tunnel to stop"
702
- return 0
703
- fi
704
-
705
- local pid
706
- pid="$(cat "${TUNNEL_PID_FILE}")"
707
- # One drain length for every caller: /suspend and /terminate always run over
708
- # a connected tunnel, and /run's and /resume's own cleanup traps only reach a
709
- # live tunnel here when it was never started (the cheap no-op branch above) —
710
- # there is no path left where a just-spawned, not-yet-connected tunnel is the
711
- # one being stopped, so there is nothing left to special-case.
712
- kill -TERM "${pid}" 2>/dev/null || true
713
-
714
- local tenths=0
715
- local deadline_tenths=$(( TUNNEL_STOP_WAIT_SECONDS * 10 ))
716
- while [ "${tenths}" -lt "${deadline_tenths}" ]; do
717
- process_is_alive "${pid}" || break
718
- sleep 0.1
719
- tenths=$(( tenths + 1 ))
720
- done
721
-
722
- # This call, not the loop's, is the deciding read: it is what the `if` below
723
- # branches on, so PROC_DEAD_REASON is fresh by construction. A corpse can
724
- # still be reaped between the loop's last poll and this call, so the reason
725
- # named below is what THIS call saw, not necessarily what the loop saw.
726
- local outcome
727
- if process_is_alive "${pid}"; then
728
- warn "tunnel ${pid} ignored SIGTERM; killing"
729
- kill -KILL "${pid}" 2>/dev/null || true
730
- outcome="after SIGKILL (ignored SIGTERM for ${TUNNEL_STOP_WAIT_SECONDS}s)"
731
- else
732
- outcome="cleanly in $(( tenths / 10 )).$(( tenths % 10 ))s (${PROC_DEAD_REASON})"
733
- fi
734
-
735
- rm -f "${TUNNEL_PID_FILE}"
736
- log "tunnel stopped ${outcome}"
737
- }
738
-
739
- # The CLI owns the interval loop and its boundary flush. Two seconds of slack
740
- # over the CLI's 8s flush deadline keeps this marker handshake inside the 55s
741
- # hook ceiling while leaving the fallback flush and tunnel stop budget intact.
742
- #
743
- # This bounds only the CLI's pre-drain flush, which always runs first and
744
- # unconditionally (run.ts's cleanup(), before the channel-work drain) — not
745
- # the CLI's post-drain second pass, which can take up to
746
- # SHUTDOWN_DRAIN_TIMEOUT_MS longer than this wait covers. A drain that
747
- # consumes the whole window loses the SECOND pass, not the credential
748
- # guarantee itself: the pre-drain flush already persisted everything on disk
749
- # at signal time, exactly what the old bash `sync_credentials` guaranteed in
750
- # one synchronous call — so the worst case here is no worse than before this
751
- # handshake existed, never a fresh data-loss window. See
752
- # docs/decisions/0063-microvm-boot-orchestration-in-cli.md's two-phase-flush
753
- # section for the full reasoning.
754
- CREDENTIAL_FLUSH_WAIT_SECONDS="${EVIDENT_CREDENTIAL_FLUSH_WAIT_SECONDS:-10}"
755
-
756
- # Remove the previous answer, signal the same CLI that stop_tunnel handles, and
757
- # wait for either its marker or its death. A fallback sync runs only after the
758
- # CLI is known to be absent, never alongside a live CLI that may still write.
759
- stop_runner_and_flush_credentials() {
760
- rm -f "${CREDENTIAL_FLUSH_MARKER_FILE}"
761
-
762
- if ! tunnel_is_running; then
763
- warn "CREDS-FLUSH-NO-RUNNER: no live CLI at handshake entry"
764
- stop_tunnel
765
- sync_credentials
766
- return 0
767
- fi
768
-
769
- local pid
770
- pid="$(cat "${TUNNEL_PID_FILE}")"
771
- kill -TERM "${pid}" 2>/dev/null || true
772
-
773
- local waited_ms=0 marker_found=false runner_exited=false
774
- while [ "${waited_ms}" -lt $((CREDENTIAL_FLUSH_WAIT_SECONDS * 1000)) ]; do
775
- if [ -s "${CREDENTIAL_FLUSH_MARKER_FILE}" ]; then
776
- marker_found=true
777
- break
778
- fi
779
- if ! process_is_alive "${pid}"; then
780
- runner_exited=true
781
- break
782
- fi
783
- sleep 0.1
784
- waited_ms=$((waited_ms + 100))
785
- done
786
-
787
- # The CLI can publish the marker and exit inside one poll tick. A final
788
- # marker check after a death break preserves that answer instead of falling
789
- # through to the fallback.
790
- if [ "${runner_exited}" = true ] && [ -s "${CREDENTIAL_FLUSH_MARKER_FILE}" ]; then
791
- marker_found=true
792
- runner_exited=false
793
- fi
794
-
795
- if [ "${marker_found}" = true ]; then
796
- local failures
797
- failures="$(grep -Ev '^(claude|opencode)=ok$' "${CREDENTIAL_FLUSH_MARKER_FILE}" || true)"
798
- if [ -n "${failures}" ]; then
799
- warn "CREDS-FLUSH-HAD-FAILURES: ${failures//$'\n'/ }"
800
- else
801
- log "CREDS-FLUSH-OK"
802
- fi
803
- stop_tunnel
804
- return 0
805
- fi
806
-
807
- if [ "${runner_exited}" = true ]; then
808
- warn "CREDS-FLUSH-RUNNER-EXITED: CLI exited without a completion marker"
809
- stop_tunnel
810
- sync_credentials
811
- return 0
812
- fi
813
-
814
- warn "CREDS-FLUSH-TIMEOUT: live CLI did not write a completion marker within ${CREDENTIAL_FLUSH_WAIT_SECONDS}s"
815
- stop_tunnel
816
- }
817
-
818
- # --- check_runner_key (#1172) ------------------------------------------------
819
- #
820
- # Answers exactly one question before opencode/the tunnel start spending this
821
- # boot's SIGTERM budget on a key that cannot work: "does the runner key in this
822
- # payload authenticate against Evident?" Delegates entirely to the CLI's own
823
- # `evident status --json` rather than
824
- # reimplementing its auth logic here — that command's `reason` field is the
825
- # published contract this function reads, and its own header states the
826
- # absent-vs-contrary distinction this function must honour.
827
- #
828
- # Keyed on the JSON `reason`, NEVER on the exit code: an older CLI without a
829
- # `status` subcommand exits 1 from Commander's own "unknown command" handling,
830
- # which under an exit-code mapping would read as "the key was rejected" and
831
- # destroy every VM on a hook/CLI version skew — precisely the hazard this
832
- # image's README's "version skew" section flags. So no parseable JSON line
833
- # means no verdict, whatever the exit code says.
834
- #
835
- # Returns 0 for positive AND absent evidence (network/timeout/5xx/404/old-CLI/
836
- # crash) — the caller continues either way — and 1 ONLY for contrary evidence
837
- # (401, another 4xx, or no credentials resolved at all): the key is actually
838
- # wrong, not merely untested (development-workflow.mdc's "never fail a gate on
839
- # absent evidence"). Every branch logs a named, greppable RUNNER-KEY-* signal
840
- # so a caller that treats the return value as fatal (or doesn't) still leaves
841
- # a trace of which branch fired.
842
- #
843
- # A 404 is deliberately NOT contrary: it means the endpoint has no /me route,
844
- # so the key was never tested. Treating it as a rejection destroyed every VM
845
- # for a day when the run payload shipped a bare origin instead of the CLI's
846
- # /v1-prefixed one.
847
- check_runner_key() {
848
- local runner_key="$1" api_url="$2"
849
-
850
- # status.ts's own exit-code contract (its header comment) means this exits
851
- # non-zero on EVERY branch except `ok`; guard the capture so an expected
852
- # non-zero does not abort this function under the caller's `set -e` before
853
- # the case below ever runs. Only stdout is
854
- # captured: status.ts's own contract is one parseable JSON line and nothing
855
- # else there, and — like litestream's stderr elsewhere in this file — its
856
- # stderr is left to reach CloudWatch directly rather than being folded in,
857
- # since a stray stderr line ahead of the JSON would otherwise break `jq`.
858
- local response rc=0
859
- response="$(EVIDENT_API_URL="${api_url}" EVIDENT_RUNNER_KEY="${runner_key}" evident status --json)" || rc=$?
860
-
861
- # `-z` as well as jq's own exit code, and not as belt-and-braces: on EMPTY
862
- # input jq has no value to report on, so `jq -e` exits 0 with empty output
863
- # rather than its documented 4 (verified on jq 1.6, the image's). Empty
864
- # stdout is exactly the old-CLI-skew and missing-binary shape this branch
865
- # exists for, so keying only on the exit code sent precisely those cases to
866
- # the unrecognised-reason arm below — same safe return, but a log line
867
- # blaming the API for an odd answer instead of naming the CLI skew, which is
868
- # the one thing an operator needs told here.
869
- local reason
870
- reason="$(printf '%s' "${response}" | jq -er '.reason' 2>/dev/null)" || reason=""
871
- if [ -z "${reason}" ]; then
872
- warn "RUNNER-KEY-UNVERIFIABLE: 'evident status --json' exited ${rc} with no parseable JSON on stdout (an old CLI without this subcommand, a missing binary, or a crash — see its stderr above, if any); the key was NOT validated, continuing anyway"
873
- return 0
874
- fi
875
- local detail
876
- detail="$(printf '%s' "${response}" | jq -er '.error // empty' 2>/dev/null)" || detail=""
877
-
878
- case "${reason}" in
879
- ok)
880
- log "RUNNER-KEY-OK: evident status confirmed the runner key against ${api_url}"
881
- return 0
882
- ;;
883
- unauthorized | no_credentials | http_error)
884
- error "RUNNER-KEY-REJECTED: evident status reason=${reason} against ${api_url}${detail:+: ${detail}}"
885
- return 1
886
- ;;
887
- unreachable)
888
- warn "RUNNER-KEY-UNREACHABLE: could not reach ${api_url} to validate the runner key; the key was NOT validated, continuing anyway"
889
- return 0
890
- ;;
891
- endpoint_not_found)
892
- # Absent evidence, NOT contrary: a 404 means this endpoint has no /me
893
- # route (typically a bare origin where the CLI wants the /v1-prefixed
894
- # one), which says nothing about the key. Loud, because the runner will
895
- # keep failing every API call until the endpoint is fixed.
896
- warn "RUNNER-KEY-ENDPOINT-NOT-FOUND: ${api_url} has no /me route${detail:+: ${detail}}; the key was NOT validated, continuing anyway"
897
- return 0
898
- ;;
899
- *)
900
- warn "RUNNER-KEY-UNKNOWN-REASON: evident status returned an unrecognised reason='${reason}'; the key was NOT validated, continuing anyway"
901
- return 0
902
- ;;
903
- esac
904
- }
905
- # --- check_runner_key (end) ---------------------------------------------------