@evident-ai/runner-cdk 3.5.2-dev.e67eb8b → 3.5.2-dev.f80111b
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +89 -39
- package/dist/controller-lambda/handler.js +16 -16
- package/dist/evident-scale-to-zero-construct.js +7 -0
- package/dist/image-version-pruner-lambda/handler.js +28250 -0
- package/dist/image-version-reporter-lambda/handler.js +15 -8
- package/dist/index.d.ts +1 -1
- package/dist/index.js +3 -2
- package/dist/microvm/constants.d.ts +1 -1
- package/dist/microvm/constants.js +7 -4
- package/dist/microvm/construct.d.ts +12 -8
- package/dist/microvm/construct.js +42 -33
- package/dist/microvm/controller/handle-doorbell.d.ts +2 -2
- package/dist/microvm/controller/handle-doorbell.js +2 -2
- package/dist/microvm/image/stage-context.d.ts +13 -38
- package/dist/microvm/image/stage-context.js +30 -120
- package/dist/microvm/image-version-pruner/construct.d.ts +13 -0
- package/dist/microvm/image-version-pruner/construct.js +79 -0
- package/dist/microvm/image-version-pruner/handler.d.ts +20 -0
- package/dist/microvm/image-version-pruner/handler.js +116 -0
- package/dist/microvm/image-version-reporter/construct.d.ts +1 -1
- package/dist/microvm/image-version-reporter/construct.js +3 -3
- package/dist/microvm/image-version-reporter/handler.d.ts +2 -2
- package/dist/microvm/image-version-reporter/handler.js +9 -7
- package/dist/microvm/shapes.d.ts +10 -21
- package/dist/microvm/shapes.js +10 -17
- package/dist/waker-lambda/handler.js +3 -3
- package/package.json +3 -3
- package/dist/microvm-image-context/Dockerfile +0 -190
- package/dist/microvm-image-context/hook-server.js +0 -286
- package/dist/microvm-image-context/hooks/common.sh +0 -904
- package/dist/microvm-image-context/hooks/resume +0 -81
- package/dist/microvm-image-context/hooks/run +0 -95
- package/dist/microvm-image-context/hooks/suspend +0 -21
- package/dist/microvm-image-context/hooks/terminate +0 -36
|
@@ -1,81 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
#
|
|
3
|
-
# Re-dial with the identity /run left behind. There is no step that could fetch
|
|
4
|
-
# a fresh runner key, so the one from /run is what resumes. The CLI delegated by
|
|
5
|
-
# start_tunnel owns the credential loop and the session-DB replicator's
|
|
6
|
-
# post-resume start.
|
|
7
|
-
set -euo pipefail
|
|
8
|
-
|
|
9
|
-
# shellcheck source=./common.sh
|
|
10
|
-
source "$(dirname "$0")/common.sh"
|
|
11
|
-
|
|
12
|
-
# Unlike /run's cleanup, this stops ONLY the tunnel: opencode is inside the
|
|
13
|
-
# snapshot and resumes with it (README.md), already serving from before this
|
|
14
|
-
# hook ever ran, and killing it here would take down a session /run started.
|
|
15
|
-
# The only failure path today is the missing-context-file check below, before
|
|
16
|
-
# start_tunnel ever runs, where stop_tunnel is a cheap no-op. Kept for the same
|
|
17
|
-
# reason /run's trap is: a future failure path introduced between start_tunnel
|
|
18
|
-
# and resume_succeeded=true must not orphan the tunnel it just spawned.
|
|
19
|
-
resume_succeeded=false
|
|
20
|
-
cleanup() {
|
|
21
|
-
if [ "${resume_succeeded}" = true ]; then
|
|
22
|
-
return 0
|
|
23
|
-
fi
|
|
24
|
-
warn "resume failed; stopping the tunnel it started"
|
|
25
|
-
stop_tunnel || warn "stop_tunnel failed while cleaning up"
|
|
26
|
-
}
|
|
27
|
-
trap cleanup EXIT
|
|
28
|
-
|
|
29
|
-
if [ ! -s "${CONTEXT_FILE}" ]; then
|
|
30
|
-
warn "resume before run: no context at ${CONTEXT_FILE}"
|
|
31
|
-
exit 1
|
|
32
|
-
fi
|
|
33
|
-
|
|
34
|
-
{
|
|
35
|
-
read -r api_url
|
|
36
|
-
read -r tunnel_url
|
|
37
|
-
read -r runner_key
|
|
38
|
-
} <"${CONTEXT_FILE}"
|
|
39
|
-
|
|
40
|
-
# Restore the id so the resumed CLI can self-report and acknowledge any
|
|
41
|
-
# outstanding recycle request (#1906). Older images may not have this file.
|
|
42
|
-
if [ -s "${MICROVM_ID_FILE}" ]; then
|
|
43
|
-
MICROVM_ID="$(cat "${MICROVM_ID_FILE}")"
|
|
44
|
-
export MICROVM_ID
|
|
45
|
-
fi
|
|
46
|
-
|
|
47
|
-
# Tolerant, never fatal: a resume that fails costs the user their whole
|
|
48
|
-
# session, so a broken durable-state config degrades to "no session-DB
|
|
49
|
-
# replication" rather than a failed resume. The CLI's own guards handle the
|
|
50
|
-
# rest — this needs no logic of its own.
|
|
51
|
-
load_state_config || warn "could not resolve durable-state config; the session DB will not resume replicating"
|
|
52
|
-
|
|
53
|
-
# Diagnostic-only, unlike /run's gate: a failed resume costs the user their
|
|
54
|
-
# whole session, so this never exits — it only converts a silent "resumed but
|
|
55
|
-
# the key is dead" into a named RUNNER-KEY-* cause in the log, whatever
|
|
56
|
-
# check_runner_key returns.
|
|
57
|
-
#
|
|
58
|
-
# And BACKGROUNDED for exactly that reason, following prewarm_litestream's
|
|
59
|
-
# pattern in common.sh: this is a network round trip bounded by the CLI's
|
|
60
|
-
# STATUS_TIMEOUT_MS (10 s), and nothing here consumes its answer. In the
|
|
61
|
-
# foreground it would sit between reading the context and `start_tunnel`,
|
|
62
|
-
# delaying the dial that IS this hook's job — by the most on exactly the
|
|
63
|
-
# unreachable API where the answer is worthless — against a resume this image
|
|
64
|
-
# measures in ~2 s (see the waker note in common.sh).
|
|
65
|
-
#
|
|
66
|
-
# Safe to leave running past the hook's own exit: the runtime waits on the
|
|
67
|
-
# script's `'exit'`, not `'close'` (aws/lambda-microvm-runtime/src/
|
|
68
|
-
# runtime.ts), so a lingering child never delays the HTTP response. Its stdio
|
|
69
|
-
# is deliberately NOT redirected — that inherited fd is how the RUNNER-KEY-*
|
|
70
|
-
# line reaches CloudWatch, which is the whole point of running it at all.
|
|
71
|
-
#
|
|
72
|
-
# The subshell is explicit rather than relying on `&` binding the whole AND-OR
|
|
73
|
-
# list, so a later edit cannot accidentally foreground the check again.
|
|
74
|
-
( check_runner_key "${runner_key}" "${api_url}" || true ) &
|
|
75
|
-
|
|
76
|
-
# Not waited on: the Evident CLI backs off and keeps retrying its own dial with no
|
|
77
|
-
# attempt cap, and Evident's control-plane reclaim pass is what decides whether a
|
|
78
|
-
# runner that never reconnects gets suspended.
|
|
79
|
-
start_tunnel "${runner_key}" "${api_url}" "${tunnel_url}"
|
|
80
|
-
|
|
81
|
-
resume_succeeded=true
|
|
@@ -1,95 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
#
|
|
3
|
-
# The per-VM phase. EVERY per-VM and secret-bearing action happens here and
|
|
4
|
-
# nowhere earlier: the image snapshot is shared by every VM started from this
|
|
5
|
-
# image version, so anything done before /run would be identical across all of
|
|
6
|
-
# them — including, fatally, a runner identity.
|
|
7
|
-
set -euo pipefail
|
|
8
|
-
|
|
9
|
-
# shellcheck source=./common.sh
|
|
10
|
-
source "$(dirname "$0")/common.sh"
|
|
11
|
-
|
|
12
|
-
# Anything this hook started must not outlive a failure: the next /run would
|
|
13
|
-
# find a stale opencode holding the port. On EXIT rather than ERR because an
|
|
14
|
-
# explicit `exit 1` — which is how every check below rejects — does NOT fire an
|
|
15
|
-
# ERR trap, and those are exactly the paths that have opencode running already.
|
|
16
|
-
# This also runs when the runtime signals an overrun hook, which it does with
|
|
17
|
-
# SIGTERM for exactly that reason: opencode and the tunnel are in a session of
|
|
18
|
-
# their own, so this trap is the only thing that can still reach them.
|
|
19
|
-
run_succeeded=false
|
|
20
|
-
cleanup() {
|
|
21
|
-
if [ "${run_succeeded}" = true ]; then
|
|
22
|
-
return 0
|
|
23
|
-
fi
|
|
24
|
-
warn "run failed; stopping what it started"
|
|
25
|
-
# Each stop tolerates its own failure: `set -e` aborts a trap function at the
|
|
26
|
-
# first non-zero command, which would skip the stop below it — precisely the
|
|
27
|
-
# orphan this trap exists to prevent.
|
|
28
|
-
stop_tunnel || warn "stop_tunnel failed while cleaning up"
|
|
29
|
-
stop_opencode || warn "stop_opencode failed while cleaning up"
|
|
30
|
-
kill_litestream || warn "kill_litestream failed while cleaning up"
|
|
31
|
-
}
|
|
32
|
-
trap cleanup EXIT
|
|
33
|
-
|
|
34
|
-
# The payload arrives on stdin, never argv.
|
|
35
|
-
payload="$(cat)"
|
|
36
|
-
|
|
37
|
-
# Never interpolate the payload into the message — it carries the runner key.
|
|
38
|
-
if ! payload_is_complete <<<"${payload}"; then
|
|
39
|
-
warn "run payload is missing runner_key, endpoints or state_prefix; refusing to dial"
|
|
40
|
-
exit 1
|
|
41
|
-
fi
|
|
42
|
-
|
|
43
|
-
# `@sh` quotes each value for the shell, so a hostile payload cannot inject.
|
|
44
|
-
eval "$(jq -er '@sh "runner_key=\(.runner_key) api_url=\(.endpoints.api) tunnel_url=\(.endpoints.tunnel) state_prefix=\(.state_prefix)"' <<<"${payload}")"
|
|
45
|
-
unset payload
|
|
46
|
-
|
|
47
|
-
# 1 — replace the identity the shared snapshot baked in.
|
|
48
|
-
regenerate_machine_id
|
|
49
|
-
|
|
50
|
-
# 2 — where this runner's credentials live. Recorded before the restore that
|
|
51
|
-
# reads it, and left behind for the hooks that flush back to the same prefix.
|
|
52
|
-
printf '%s\n' "${state_prefix}" >"${STATE_PREFIX_FILE}"
|
|
53
|
-
|
|
54
|
-
# 3 — start reading the litestream and aws CLI binaries NOW. The reads run in
|
|
55
|
-
# the background and overlap the CLI's own node startup and auth round trip,
|
|
56
|
-
# keeping their cold first touches off the path to the first credential fetch.
|
|
57
|
-
prewarm_litestream
|
|
58
|
-
prewarm_aws_cli
|
|
59
|
-
|
|
60
|
-
# 4 — export the durable-state config before anything that reads it. Fatal,
|
|
61
|
-
# because the CLI and the teardown hooks must agree on this runner's state prefix.
|
|
62
|
-
load_state_config || exit 1
|
|
63
|
-
|
|
64
|
-
# 5 — does the runner key in this payload actually authenticate against
|
|
65
|
-
# Evident? Checked here, before the tunnel spends any of this boot's time on a key
|
|
66
|
-
# that cannot work, and before anything has started that cleanup would need to
|
|
67
|
-
# tear down. Fatal ONLY on contrary evidence
|
|
68
|
-
# (check_runner_key's own contract, common.sh): a key the API actively rejects
|
|
69
|
-
# cannot work regardless, and failing here costs ~5s against the ~10 minutes a
|
|
70
|
-
# runner that can never connect would otherwise burn before the lifecycle cron
|
|
71
|
-
# reclaims it. Absent evidence (unreachable, no parseable JSON) is not fatal —
|
|
72
|
-
# it warns and this VM still boots.
|
|
73
|
-
check_runner_key "${runner_key}" "${api_url}" || exit 1
|
|
74
|
-
|
|
75
|
-
# 6 — preserve the VM id across suspend/resume. The runtime always supplies it
|
|
76
|
-
# for a /run hook, and the subshell keeps the persisted value protected.
|
|
77
|
-
(
|
|
78
|
-
umask 077
|
|
79
|
-
printf '%s\n' "${MICROVM_ID}" >"${MICROVM_ID_FILE}"
|
|
80
|
-
)
|
|
81
|
-
|
|
82
|
-
# 7 — the first per-VM identity on the wire. The subshell's umask makes the file
|
|
83
|
-
# unreadable to anyone else from the moment it exists, before the key is in it.
|
|
84
|
-
(
|
|
85
|
-
umask 077
|
|
86
|
-
printf '%s\n%s\n%s\n' "${api_url}" "${tunnel_url}" "${runner_key}" \
|
|
87
|
-
>"${CONTEXT_FILE}"
|
|
88
|
-
)
|
|
89
|
-
|
|
90
|
-
# Not waited on: the Evident CLI backs off and keeps retrying its own dial with no
|
|
91
|
-
# attempt cap, and Evident's control-plane reclaim pass is what decides whether a
|
|
92
|
-
# runner that never connects gets suspended.
|
|
93
|
-
start_tunnel "${runner_key}" "${api_url}" "${tunnel_url}" true true
|
|
94
|
-
|
|
95
|
-
run_succeeded=true
|
|
@@ -1,21 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
#
|
|
3
|
-
# Flush anything durable, then drop the tunnel so the relay stops routing to a
|
|
4
|
-
# VM that is about to freeze. opencode stays up: it is inside the snapshot and
|
|
5
|
-
# resumes with it.
|
|
6
|
-
set -euo pipefail
|
|
7
|
-
|
|
8
|
-
# shellcheck source=./common.sh
|
|
9
|
-
source "$(dirname "$0")/common.sh"
|
|
10
|
-
|
|
11
|
-
# load_state_config exports PERSISTENCE_BUCKET, which flush_session_db below
|
|
12
|
-
# requires; the CLI flush completes, or is proven absent, before teardown continues.
|
|
13
|
-
load_state_config || warn "could not resolve durable-state config; neither the credential flush nor the session-DB flush below can run"
|
|
14
|
-
stop_runner_and_flush_credentials
|
|
15
|
-
|
|
16
|
-
# opencode is deliberately NOT stopped here (#812 WI-4) — it is inside the
|
|
17
|
-
# snapshot and must resume with it — so a turn still in flight may not be
|
|
18
|
-
# fully flushed by this call. Acceptable: the VM resumes with the local DB
|
|
19
|
-
# intact, and this flush exists for the TERMINATED-while-suspended case,
|
|
20
|
-
# where a frozen VM would otherwise flush nothing at all.
|
|
21
|
-
flush_session_db
|
|
@@ -1,36 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
#
|
|
3
|
-
# Best-effort teardown. The runtime reports 200 whatever happens here — the VM
|
|
4
|
-
# is going away either way — so each step is allowed to fail on its own without
|
|
5
|
-
# skipping the ones after it.
|
|
6
|
-
set -uo pipefail
|
|
7
|
-
|
|
8
|
-
# shellcheck source=./common.sh
|
|
9
|
-
source "$(dirname "$0")/common.sh"
|
|
10
|
-
|
|
11
|
-
# load_state_config exports PERSISTENCE_BUCKET, which flush_session_db below
|
|
12
|
-
# requires; the CLI flush completes, or is proven absent, before teardown continues.
|
|
13
|
-
load_state_config || warn "could not resolve durable-state config; neither the credential flush nor the session-DB flush below can run"
|
|
14
|
-
stop_runner_and_flush_credentials
|
|
15
|
-
|
|
16
|
-
# Writers stopped BEFORE litestream's final sync (#812 WI-4, the
|
|
17
|
-
# ordered-shutdown invariant, plan §6): otherwise litestream could snapshot
|
|
18
|
-
# while opencode is still writing its WAL and the last session writes would
|
|
19
|
-
# be missing from S3. stop_opencode_and_wait, not /run cleanup's cheap
|
|
20
|
-
# stop_opencode — /terminate can afford the real drain, the VM is being torn
|
|
21
|
-
# down either way.
|
|
22
|
-
stop_opencode_and_wait
|
|
23
|
-
flush_session_db
|
|
24
|
-
|
|
25
|
-
# The key must not outlive the VM's last useful moment. The no-replicate
|
|
26
|
-
# marker goes alongside it: no state that a later /resume could trust should
|
|
27
|
-
# outlive the VM's last useful moment either.
|
|
28
|
-
#
|
|
29
|
-
# The tunnel-ready marker used to be removed here too. It is gone entirely
|
|
30
|
-
# since #1172 — and naming an unset variable here is not a harmless leftover:
|
|
31
|
-
# this hook runs under `set -u` (above), so an unbound one aborts the script
|
|
32
|
-
# AT THIS LINE, leaving the runner key sitting in CONTEXT_FILE on a VM that is
|
|
33
|
-
# being torn down. That is the one thing this line exists to prevent.
|
|
34
|
-
rm -f "${CONTEXT_FILE}" "${SESSION_DB_NO_REPLICATE_MARKER}" "${CREDENTIAL_FLUSH_MARKER_FILE}"
|
|
35
|
-
|
|
36
|
-
exit 0
|