kampodine 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -3
- package/package.json +1 -1
- package/scripts/bluegreen.sh +439 -53
- package/scripts/image-import.sh +41 -10
- package/scripts/migrate.sh +32 -6
- package/scripts/vm-prepare.sh +180 -7
package/README.md
CHANGED
|
@@ -50,10 +50,21 @@ kampodine deploy:
|
|
|
50
50
|
previous image stays on the VM for exactly this)
|
|
51
51
|
- **`kampodine bluegreen status|init|provision|flip|rollback`** — reserved
|
|
52
52
|
public IP management for a two-instance blue/green pair (zero DNS change,
|
|
53
|
-
health-gated flips, auto-rollback)
|
|
53
|
+
health-gated flips, auto-rollback). **0.2.0: live-proven end-to-end** —
|
|
54
|
+
provision via platform-launch + golden-qcow2 disk injection (~9 min to a
|
|
55
|
+
booting Alpine A1; OCI custom-image import is BIOS-pinned and A1-rejects),
|
|
56
|
+
a guest **anchor watcher** shipped by `vm-prepare` (add-only, inert until
|
|
57
|
+
a flip writes `/etc/esellar/anchor.conf`), the **ACME-first flip**
|
|
58
|
+
(~27 s: anchor → OCI assign → cert for the reserved-IP hostname →
|
|
59
|
+
served-sha verify) with a ~6 s rollback, and a strict inject gate
|
|
60
|
+
(release must match Alpine `3.x`; reboot failures die loudly)
|
|
54
61
|
- **`kampodine vm-prepare`** — first-run bootstrap of a bare Alpine host
|
|
55
|
-
-
|
|
56
|
-
|
|
62
|
+
(0.2.0: sshd-hardening self-heal — fresh VMs pass unattended; busybox-wget
|
|
63
|
+
probes — the golden image ships no curl; ships `esellar-anchor`)
|
|
64
|
+
- **`kampodine image-import`** — golden qcow2 → OCI custom image (0.2.0:
|
|
65
|
+
every `oci` call honors `OCI_PROFILE`; note A1 rejects imported images —
|
|
66
|
+
use `bluegreen provision` for Ampere targets)
|
|
67
|
+
- **`kampodine migrate`** — sqlite migrations overSSH
|
|
57
68
|
|
|
58
69
|
## Kamal parity
|
|
59
70
|
|
package/package.json
CHANGED
package/scripts/bluegreen.sh
CHANGED
|
@@ -13,11 +13,59 @@
|
|
|
13
13
|
# Usage:
|
|
14
14
|
# kampodine bluegreen status # pair view: instances, IP holder, health
|
|
15
15
|
# kampodine bluegreen init # create the reserved public IP (dormant, unassigned)
|
|
16
|
-
# kampodine bluegreen provision <color> # launch the second instance
|
|
17
|
-
# kampodine bluegreen flip --to <color> # health-gated
|
|
18
|
-
# kampodine bluegreen rollback #
|
|
16
|
+
# kampodine bluegreen provision <color> # launch the second instance (golden image; see below)
|
|
17
|
+
# kampodine bluegreen flip --to <color> # ACME-first health-gated flip (see below)
|
|
18
|
+
# kampodine bluegreen rollback # unassign to DORMANT + holder guest cleanup
|
|
19
|
+
#
|
|
20
|
+
# FLIP = ACME-FIRST (closes the 2026-10-07 pt2 serving-leg gap):
|
|
21
|
+
# 1. health gate on the target's own IP
|
|
22
|
+
# 2. anchor.conf written on the TARGET guest over ssh — the esellar-anchor
|
|
23
|
+
# watcher service (shipped by vm-prepare) configures the anchor private
|
|
24
|
+
# address within one interval; flip waits for `ip addr` to show it
|
|
25
|
+
# 3. OCI assigns the reserved IP to the target's anchor (secondary private
|
|
26
|
+
# ip — the primary holds the launch-time ephemeral, 409 live-proven)
|
|
27
|
+
# 4. ACME on the TARGET's kamal-proxy for the reserved-IP sslip hostname
|
|
28
|
+
# (`podman exec kamal-proxy kamal-proxy deploy --host=<raddr-dashes>
|
|
29
|
+
# .sslip.io --tls`): HTTP-01 needs the hostname to already resolve to
|
|
30
|
+
# the reserved ip AND route to the target — hence AFTER the assign, and
|
|
31
|
+
# hence NOT a deploy-time cert (a deploy before the flip cannot issue
|
|
32
|
+
# for the reserved hostname, and issuing on every deploy would burn LE
|
|
33
|
+
# rate limits for VMs that never flip). kamal-proxy has no standalone
|
|
34
|
+
# issue verb — `deploy` IS its registration+ACME path, the same
|
|
35
|
+
# invocation `kampodine deploy` uses; certs persist in the
|
|
36
|
+
# kamal-proxy-config volume. sslip.io sits on the public suffix list, so
|
|
37
|
+
# the reserved-hostname quota is per-hostname.
|
|
38
|
+
# 5. verify https through the reserved ip (curl --resolve, valid cert for
|
|
39
|
+
# the sslip name + /up 200) and the served sha — only then: FLIPPED.
|
|
40
|
+
# Any post-assign failure auto-rolls back: the reserved ip goes to the
|
|
41
|
+
# other color's EXISTING anchor (lookup-ONLY — drills never mint green-side
|
|
42
|
+
# artifacts) or, when none exists, back to UNASSIGNED/dormant; the failed
|
|
43
|
+
# target's anchor.conf is removed and its anchor address deleted.
|
|
44
|
+
#
|
|
45
|
+
# ROLLBACK = back to dormant: holder cleanup (anchor.conf removal + address
|
|
46
|
+
# delete over ssh) then the OCI unassign (`--private-ip-id ""`). For a real
|
|
47
|
+
# pair where traffic must move to the other color, use `flip --to <other>`
|
|
48
|
+
# instead — it runs the same ACME-first sequence there.
|
|
49
|
+
#
|
|
50
|
+
# provision has TWO routes:
|
|
51
|
+
# 1. NATIVE — a UEFI_64 esellar-alpine* custom image exists in the
|
|
52
|
+
# compartment: launch it directly (the golden image boots as-is).
|
|
53
|
+
# 2. INJECT (provision-via-migrate) — OCI pins imported custom images to
|
|
54
|
+
# firmware=BIOS and A1/Ampere is UEFI-only (exhaustively proven,
|
|
55
|
+
# RUNBOOK §Blue-green), but green itself runs Alpine on a boot volume
|
|
56
|
+
# whose image metadata is the Ubuntu PLATFORM image: green was built by
|
|
57
|
+
# platform-image launch + disk injection. With no UEFI custom image,
|
|
58
|
+
# provision queries the template instance's LIVE image-id (never
|
|
59
|
+
# hardcoded — proven A1-launchable, since the template runs on it),
|
|
60
|
+
# launches from it with the ops ssh key, then streams the golden qcow2
|
|
61
|
+
# onto the new instance's boot disk (qemu-img convert -> gzip | ssh
|
|
62
|
+
# 'gunzip | sudo dd', reboot, verify /etc/alpine-release). The instance
|
|
63
|
+
# record keeps the platform image metadata — exactly like green.
|
|
19
64
|
#
|
|
20
65
|
# Env: OCI_PROFILE (default esellar-api), OCI_COMPARTMENT (default esellar).
|
|
66
|
+
# Injection extras: ALPINE_QCOW2 (golden disk path), OPS_SSH_PUBKEY (ops
|
|
67
|
+
# public key for the platform-image launch), PLATFORM_SSH_USER (default
|
|
68
|
+
# ubuntu), INJECT_PROBE_SLEEP / INJECT_PROBE_TRIES (ssh wait tuning).
|
|
21
69
|
# Health checks SSH to each instance's OWN ephemeral IP (the app is checked on
|
|
22
70
|
# loopback; the reserved IP is checked over :80 with the prod Host header).
|
|
23
71
|
set -euo pipefail
|
|
@@ -25,6 +73,7 @@ set -euo pipefail
|
|
|
25
73
|
PROFILE="${OCI_PROFILE:-esellar-api}"
|
|
26
74
|
COMPARTMENT_NAME="${OCI_COMPARTMENT:-esellar}"
|
|
27
75
|
APP_HOST_HEADER="${APP_HOST_HEADER:-84-13-128-216.sslip.io}"
|
|
76
|
+
REPO_ROOT="${REPO_ROOT:-$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd)}"
|
|
28
77
|
SSH_OPTS=(-o ConnectTimeout=6 -o BatchMode=yes -o StrictHostKeyChecking=accept-new)
|
|
29
78
|
|
|
30
79
|
say() { printf '%s\n' "$*"; }
|
|
@@ -67,27 +116,244 @@ reserved_ip() {
|
|
|
67
116
|
printf '%s' "$row"
|
|
68
117
|
}
|
|
69
118
|
|
|
70
|
-
#
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
119
|
+
# primary_vnic_of <instance_ocid> -> primary VNIC ocid (vnic-attachment list
|
|
120
|
+
# REQUIRES the compartment flag — reverse of private-ip list, which rejects
|
|
121
|
+
# it; both pinned in the stub tests)
|
|
122
|
+
primary_vnic_of() {
|
|
123
|
+
local iid="$1"
|
|
124
|
+
oci compute vnic-attachment list -c "$(compartment_ocid)" --profile "$PROFILE" \
|
|
125
|
+
--instance-id "$iid" --query 'data[0]."vnic-id"' --raw-output 2>/dev/null || true
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
# reserved_anchor_ip <vnic_ocid> -> private-ip ocid to anchor the reserved
|
|
129
|
+
# public ip on. OCI allows ONE public ip per private ip: the PRIMARY private
|
|
130
|
+
# ip holds the instance's launch-time EPHEMERAL public ip, so assigning the
|
|
131
|
+
# reserved there is a 409 Conflict ("already has a public IP", live-proven
|
|
132
|
+
# 2026-10-07). The reserved anchors on a SECONDARY private ip instead: both
|
|
133
|
+
# colors keep their ephemeral address + reachability, the flip moves only
|
|
134
|
+
# the reserved ip between secondaries.
|
|
135
|
+
# existing_anchor_ip <vnic> -> the SECONDARY private-ip ocid or empty.
|
|
136
|
+
# LOOKUP-ONLY: auto-rollback/rollback resolve holders with this — they must
|
|
137
|
+
# never create artifacts on a color they are merely pointing traffic at.
|
|
138
|
+
existing_anchor_ip() {
|
|
139
|
+
local vnic="$1"
|
|
140
|
+
oci network private-ip list --profile "$PROFILE" --vnic-id "$vnic" \
|
|
141
|
+
--query 'data[? "is-primary" == `false` ] | [0].id' --raw-output 2>/dev/null || true
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
# reserved_anchor_ip <vnic> -> the anchor ocid, creating it when absent (the
|
|
145
|
+
# flip TARGET is the only legitimate creation site).
|
|
146
|
+
reserved_anchor_ip() {
|
|
147
|
+
local vnic="$1" sec
|
|
148
|
+
sec="$(existing_anchor_ip "$vnic")"
|
|
149
|
+
if [[ "$sec" == ocid1.privateip* ]]; then
|
|
150
|
+
printf '%s' "$sec"
|
|
151
|
+
return 0
|
|
152
|
+
fi
|
|
153
|
+
oci network private-ip create --profile "$PROFILE" --vnic-id "$vnic" \
|
|
154
|
+
--display-name "esellar-reserved-anchor" \
|
|
155
|
+
--query 'data.id' --raw-output 2>/dev/null
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
# Flip-time pacing (tests run with FLIP_POLL_SLEEP=0).
|
|
159
|
+
FLIP_POLL_SLEEP="${FLIP_POLL_SLEEP:-5}"
|
|
160
|
+
FLIP_ADDR_TRIES="${FLIP_ADDR_TRIES:-12}" # guest watcher pickup: 12 x 5s = 60s
|
|
161
|
+
FLIP_ACME_TRIES="${FLIP_ACME_TRIES:-24}" # LE HTTP-01: 24 x 5s = 120s
|
|
162
|
+
|
|
163
|
+
# Guest anchor protocol (the esellar-anchor watcher half lives in
|
|
164
|
+
# infra/alpine-host/ansible/roles/container-service/files/esellar-anchor.sh —
|
|
165
|
+
# both halves of the conf path must stay in sync: /etc/esellar/anchor.conf).
|
|
166
|
+
# Remote commands are composed client-side BY DESIGN (vm-prepare convention);
|
|
167
|
+
# the interpolated values are OCI-API derived and regex-gated at the call
|
|
168
|
+
# sites, never user input.
|
|
169
|
+
|
|
170
|
+
# shellcheck disable=SC2029
|
|
171
|
+
write_anchor_conf() {
|
|
172
|
+
local ip="$1" addr="$2"
|
|
173
|
+
ssh "${SSH_OPTS[@]}" "root@$ip" \
|
|
174
|
+
"umask 077; mkdir -p /etc/esellar; printf 'ANCHOR_ADDR=%s\nANCHOR_IFACE=\n' '$addr' > /etc/esellar/anchor.conf && echo ANCHOR_CONF_WRITTEN"
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
# shellcheck disable=SC2029
|
|
178
|
+
anchor_addr_ready() {
|
|
179
|
+
local ip="$1" addr="$2"
|
|
180
|
+
ssh "${SSH_OPTS[@]}" "root@$ip" "ip -4 addr show | grep -qF -- '$addr'"
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
# read_anchor_conf_addr <ip> -> the guest's current ANCHOR_ADDR (with prefix)
|
|
184
|
+
# or empty. Quotes stripped — the conf is shell-sourceable by the watcher.
|
|
185
|
+
read_anchor_conf_addr() {
|
|
186
|
+
local ip="$1" line addr
|
|
187
|
+
line="$(ssh "${SSH_OPTS[@]}" "root@$ip" \
|
|
188
|
+
'grep -h "^ANCHOR_ADDR=" /etc/esellar/anchor.conf 2>/dev/null | head -n 1' 2>/dev/null || true)"
|
|
189
|
+
addr="${line#ANCHOR_ADDR=}"
|
|
190
|
+
addr="${addr//\"/}"
|
|
191
|
+
addr="${addr//\'/}"
|
|
192
|
+
[[ "$addr" == */* ]] && printf '%s' "$addr"
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
# delete_guest_addr <ip> <addr> — the flip tool's explicit cleanup (the
|
|
196
|
+
# WATCHER is add-only by design; removal is always ours, after the conf is
|
|
197
|
+
# gone so it cannot be re-added).
|
|
198
|
+
# shellcheck disable=SC2029
|
|
199
|
+
delete_guest_addr() {
|
|
200
|
+
local ip="$1" addr="$2"
|
|
201
|
+
ssh "${SSH_OPTS[@]}" "root@$ip" \
|
|
202
|
+
"iface=\$(ip -4 route show default 2>/dev/null | awk '{print \$5; exit}'); [ -n \"\$iface\" ] && ip addr del '$addr' dev \"\$iface\" 2>/dev/null; true" \
|
|
203
|
+
>/dev/null 2>&1 || true
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
# cleanup_target_anchor <ip> <addr_with_prefix> — conf removal + address
|
|
207
|
+
# deletion. The watcher may have sourced the conf just before removal and
|
|
208
|
+
# re-added the address once; retry the delete across one watcher interval.
|
|
209
|
+
cleanup_target_anchor() {
|
|
210
|
+
local ip="$1" addr="$2" i
|
|
211
|
+
ssh "${SSH_OPTS[@]}" "root@$ip" "rm -f /etc/esellar/anchor.conf" >/dev/null 2>&1 || true
|
|
212
|
+
for ((i = 1; i <= 3; i++)); do
|
|
213
|
+
delete_guest_addr "$ip" "$addr"
|
|
214
|
+
anchor_addr_ready "$ip" "$addr" >/dev/null 2>&1 || return 0
|
|
215
|
+
sleep "$((FLIP_POLL_SLEEP * 2))"
|
|
216
|
+
done
|
|
217
|
+
say "warning: guest anchor address $addr still present on $ip after cleanup (watcher resurrection?) — inert once the reserved ip moves, verify before reuse" >&2
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
# flip_failure_rollback <failed_color> <other_color> <comp> <reserved_ocid> <target_ip> <target_addr>
|
|
221
|
+
# Post-assign failure path: reassign the reserved ip to the other color's
|
|
222
|
+
# EXISTING anchor (lookup-ONLY — auto-rollback must never mint artifacts on
|
|
223
|
+
# the holder we are failing away from), else unassign to dormant; then clean
|
|
224
|
+
# the failed target's guest state.
|
|
225
|
+
flip_failure_rollback() {
|
|
226
|
+
local to="$1" ob="$2" comp="$3" rocid="$4" tpub="$5" taddr="$6"
|
|
227
|
+
local recovered=0 brow bnic bpip
|
|
228
|
+
say "post-assign failure — auto-rollback" >&2
|
|
229
|
+
brow="$(instance_by_color "$comp" "$ob")"
|
|
230
|
+
if [[ -n "$brow" ]]; then
|
|
231
|
+
bnic="$(primary_vnic_of "$(cut -d' ' -f1 <<<"$brow")")"
|
|
232
|
+
bpip=""
|
|
233
|
+
[[ -n "$bnic" ]] && bpip="$(existing_anchor_ip "$bnic")"
|
|
234
|
+
if [[ "$bpip" == ocid1.privateip* ]]; then
|
|
235
|
+
if oci network public-ip update --public-ip-id "$rocid" --profile "$PROFILE" \
|
|
236
|
+
--private-ip-id "$bpip" --force >/dev/null 2>&1; then
|
|
237
|
+
say "rolled back to esellar-$ob (existing anchor $bpip)" >&2
|
|
238
|
+
recovered=1
|
|
239
|
+
fi
|
|
240
|
+
fi
|
|
241
|
+
fi
|
|
242
|
+
if (( ! recovered )); then
|
|
243
|
+
if oci network public-ip update --public-ip-id "$rocid" --profile "$PROFILE" \
|
|
244
|
+
--private-ip-id "" --force --wait-for-state AVAILABLE >/dev/null 2>&1; then
|
|
245
|
+
say "reserved ip UNASSIGNED (dormant) — no existing anchor on esellar-$ob" >&2
|
|
246
|
+
else
|
|
247
|
+
cleanup_target_anchor "$tpub" "$taddr"
|
|
248
|
+
die "AUTO-ROLLBACK FAILED — reserved ip state unknown; flip manually via console: $rocid"
|
|
249
|
+
fi
|
|
250
|
+
fi
|
|
251
|
+
cleanup_target_anchor "$tpub" "$taddr"
|
|
252
|
+
say "esellar-$to cleaned (anchor.conf removed, anchor address deleted) — investigate before retrying" >&2
|
|
78
253
|
}
|
|
79
254
|
|
|
80
|
-
# instance_healthy <ephemeral_ip> -> ssh + loopback app check
|
|
255
|
+
# instance_healthy <ephemeral_ip> -> ssh + loopback app check.
|
|
256
|
+
# busybox wget, NOT curl: the golden image is deliberately minimal (ssh +
|
|
257
|
+
# OpenRC + busybox) and ships no curl (2026-10-07 flip attempt 3: the gate
|
|
258
|
+
# failed "unhealthy" on a healthy blue because curl is absent).
|
|
81
259
|
instance_healthy() {
|
|
82
260
|
local ip="$1"
|
|
83
261
|
[[ -n "$ip" ]] || return 1
|
|
84
262
|
ssh "${SSH_OPTS[@]}" "root@$ip" \
|
|
85
|
-
"rc-service esellar-api status >/dev/null 2>&1 &&
|
|
263
|
+
"rc-service esellar-api status >/dev/null 2>&1 && busybox wget -q -O /dev/null http://127.0.0.1:8080/up" 2>/dev/null
|
|
86
264
|
}
|
|
87
265
|
|
|
88
|
-
#
|
|
89
|
-
|
|
90
|
-
|
|
266
|
+
# instance_wait_running <iid> — poll lifecycle-state to RUNNING. The OCI CLI's
|
|
267
|
+
# `instance get` has NO --wait-for-state option (3.94.1: "No such option" —
|
|
268
|
+
# the 2026-10-07 drill's first provision died on exactly that), so poll.
|
|
269
|
+
# Dies on the terminal-bad states; everything else (PROVISIONING, STARTING…)
|
|
270
|
+
# keeps the loop going.
|
|
271
|
+
instance_wait_running() {
|
|
272
|
+
local iid="$1" sleep_s="${INJECT_PROBE_SLEEP:-5}" tries="${WAIT_RUNNING_TRIES:-120}"
|
|
273
|
+
local state="" i
|
|
274
|
+
for ((i = 1; i <= tries; i++)); do
|
|
275
|
+
state="$(oci compute instance get --instance-id "$iid" --profile "$PROFILE" \
|
|
276
|
+
--query 'data."lifecycle-state"' --raw-output 2>/dev/null || true)"
|
|
277
|
+
case "$state" in
|
|
278
|
+
RUNNING) return 0 ;;
|
|
279
|
+
FAILED | TERMINATED | TERMINATING) die "esellar instance reached $state — nothing to inject, check the console" ;;
|
|
280
|
+
esac
|
|
281
|
+
sleep "$sleep_s"
|
|
282
|
+
done
|
|
283
|
+
die "instance never reached RUNNING (last state: ${state:-unknown}) after $((tries * 10))s"
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
# inject_alpine <color> <ip> <qcow2> — provision-via-migrate: stream the golden
|
|
287
|
+
# Alpine disk onto a RUNNING platform-image instance's boot volume. The disk
|
|
288
|
+
# write that built green: qcow2 -> raw locally, gzip | ssh 'gunzip | sudo dd'
|
|
289
|
+
# over the ssh-detected boot disk (lsblk PKNAME of the / mount — never
|
|
290
|
+
# hardcoded), conv=fsync, `reboot -f` (the fs it would unmount is gone), then
|
|
291
|
+
# verify Alpine answers on ssh. Instance + boot-volume image metadata stay the
|
|
292
|
+
# platform image — exactly like green.
|
|
293
|
+
inject_alpine() {
|
|
294
|
+
local color="$1" ip="$2" qcow2="$3"
|
|
295
|
+
local probe_sleep="${INJECT_PROBE_SLEEP:-5}" probe_tries="${INJECT_PROBE_TRIES:-60}"
|
|
296
|
+
local ruser="${PLATFORM_SSH_USER:-ubuntu}" rel="" raw kh
|
|
297
|
+
# Dedicated throwaway known-hosts for the WHOLE injection phase: the same
|
|
298
|
+
# IP runs sshd on TWO different host keys (Ubuntu platform first boot ->
|
|
299
|
+
# injected Alpine first boot). accept-new takes the first, then refuses
|
|
300
|
+
# the CHANGED key after the reboot — the user's known_hosts would wedge
|
|
301
|
+
# the Alpine verify forever (2026-10-07 live drill). The scrub below
|
|
302
|
+
# clears the phase file between the two boots; the user's file is never
|
|
303
|
+
# touched.
|
|
304
|
+
kh="$(mktemp "${TMPDIR:-/tmp}/esellar-inject-kh.XXXXXX")"
|
|
305
|
+
iss() { ssh -o UserKnownHostsFile="$kh" "${SSH_OPTS[@]}" "$@"; }
|
|
306
|
+
say "inject: waiting for ssh (${ruser}@${ip}, platform-image first boot)…"
|
|
307
|
+
ssh_wait_probe() { iss "${ruser}@${ip}" true; }
|
|
308
|
+
local i
|
|
309
|
+
for ((i = 1; i <= probe_tries; i++)); do
|
|
310
|
+
ssh_wait_probe >/dev/null 2>&1 && break
|
|
311
|
+
[[ "$i" == "$probe_tries" ]] && die "ssh never came up on ${ip} (${ruser}) — check the instance console connection"
|
|
312
|
+
sleep "$probe_sleep"
|
|
313
|
+
done
|
|
314
|
+
raw="$(mktemp "${TMPDIR:-/tmp}/esellar-inject-raw.XXXXXX")"
|
|
315
|
+
say "inject: converting ${qcow2} -> raw…"
|
|
316
|
+
qemu-img convert -O raw "$qcow2" "$raw"
|
|
317
|
+
say "inject: streaming golden disk -> ${ip} boot volume (gunzip | dd, conv=fsync)…"
|
|
318
|
+
if ! gzip -c "$raw" | iss "${ruser}@${ip}" \
|
|
319
|
+
'set -eu; DISK="$(lsblk -no PKNAME "$(findmnt -n -o SOURCE /)")"; [ -n "$DISK" ] || exit 3; echo "[inject] writing /dev/$DISK"; gunzip -c | sudo dd of="/dev/$DISK" bs=4M conv=fsync status=progress'; then
|
|
320
|
+
rm -f "$raw" "$kh"
|
|
321
|
+
die "disk stream to ${ip} failed — instance left UNBOOTABLE-ish (platform image partially overwritten): terminate it, do NOT flip to ${color}"
|
|
322
|
+
fi
|
|
323
|
+
rm -f "$raw"
|
|
324
|
+
# ClientAlive keepalives on the reboot call: reboot -f kills the platform
|
|
325
|
+
# sshd WITHOUT closing the TCP session, and ConnectTimeout only bounds
|
|
326
|
+
# connection ESTABLISHMENT — the pt3 drill hung the whole provisioner on
|
|
327
|
+
# that wedged session (the guest was long up when the client gave up).
|
|
328
|
+
# ClientAliveCountMax x Interval bounds a dead session to ~15s.
|
|
329
|
+
say "inject: rebooting into the injected disk (reboot -f — the old fs is gone)…"
|
|
330
|
+
# Surface the reboot failure: `|| true` here once false-INJECTED (Phase-4
|
|
331
|
+
# rehearsal finding 2026-10-07) — the wedged Ubuntu guest kept answering
|
|
332
|
+
# ssh and the probe below accepted its banner as "Alpine boots". The
|
|
333
|
+
# reboot must SUCCEED for the inject to be real (the disk was replaced).
|
|
334
|
+
if ! iss -o ClientAliveInterval=5 -o ClientAliveCountMax=3 "${ruser}@${ip}" 'sudo reboot -f' >/dev/null 2>&1; then
|
|
335
|
+
die "inject: reboot -f FAILED on ${ip} — the injection did not take (terminate, do NOT flip to ${color})"
|
|
336
|
+
fi
|
|
337
|
+
# the injected Alpine boots a NEW host key under the SAME ip — scrub the
|
|
338
|
+
# phase file so the probes below see it as a fresh accept-new
|
|
339
|
+
ssh-keygen -R "$ip" -f "$kh" >/dev/null 2>&1 || true
|
|
340
|
+
say "inject: waiting for Alpine ssh (root@${ip})…"
|
|
341
|
+
rel=""
|
|
342
|
+
for ((i = 1; i <= probe_tries; i++)); do
|
|
343
|
+
rel="$(iss "root@${ip}" 'cat /etc/alpine-release' 2>/dev/null || true)"
|
|
344
|
+
# Verify the RELEASE STRING, not non-empty output: the Phase-4 rehearsal
|
|
345
|
+
# false-INJECTED when the un-rebooted Ubuntu guest's ssh banner satisfied
|
|
346
|
+
# a non-empty check. /etc/alpine-release only exists on a real Alpine
|
|
347
|
+
# boot and reads 3.x for every image we ship.
|
|
348
|
+
if [[ "$rel" =~ ^3\.[0-9]+\.[0-9]+ ]]; then break; fi
|
|
349
|
+
rel=""
|
|
350
|
+
sleep "$probe_sleep"
|
|
351
|
+
done
|
|
352
|
+
rm -f "$kh"
|
|
353
|
+
[[ -n "$rel" ]] || die "injection streamed but no ALPINE 3.x ssh on ${ip} after reboot (got: '${rel:-nothing}') — check the serial console; terminate, do NOT flip to ${color}"
|
|
354
|
+
say "INJECTED esellar-$color: Alpine ${rel} boots on ${ip} (instance image metadata stays the platform image — like green)"
|
|
355
|
+
say "next: kampodine vm-prepare --host root@${ip} -> kampodine deploy --host root@${ip}"
|
|
356
|
+
say "then 'bluegreen.sh flip --to ${color}' (health-gated) once its app checks green."
|
|
91
357
|
}
|
|
92
358
|
|
|
93
359
|
cmd="${1:-}"
|
|
@@ -140,22 +406,74 @@ case "$cmd" in
|
|
|
140
406
|
other=green; [[ "$color" == green ]] && other=blue
|
|
141
407
|
orow="$(instance_by_color "$comp" "$other")"
|
|
142
408
|
[[ -n "$orow" ]] || die "esellar-$other not RUNNING — need its AD/subnet as the pair template"
|
|
143
|
-
|
|
409
|
+
# instance_by_color rows are "<ocid> <ephemeral_public_ip> <ad>" — field 2
|
|
410
|
+
# is the IP, the AD is field 3 (launching with the IP as AD fails).
|
|
411
|
+
read -r oiid _ oad <<<"$orow"
|
|
144
412
|
subnet="$(oci compute vnic-attachment list -c "$comp" --profile "$PROFILE" \
|
|
145
413
|
--instance-id "$oiid" --query 'data[0]."subnet-id"' --raw-output)"
|
|
146
|
-
|
|
414
|
+
|
|
415
|
+
# Route selection. NATIVE only with a UEFI_64 esellar-alpine* custom image:
|
|
416
|
+
# OCI pins IMPORTED images to firmware=BIOS and A1 is UEFI-only, so a BIOS
|
|
417
|
+
# verdict means the import would die at launch (Shape ... is not valid for
|
|
418
|
+
# image) — skip it. Otherwise take the template's LIVE image-id from its
|
|
419
|
+
# instance record and inject the golden disk (provision-via-migrate).
|
|
420
|
+
mode="" image=""
|
|
421
|
+
custom="$(oci compute image list -c "$comp" --profile "$PROFILE" --all --sort-by TIMECREATED \
|
|
147
422
|
--query "data[?\"display-name\" != null && starts_with(\"display-name\", 'esellar-alpine')] | [0].id" \
|
|
148
423
|
--raw-output 2>/dev/null || true)"
|
|
149
|
-
[[ "$
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
424
|
+
if [[ "$custom" == ocid1.image* ]]; then
|
|
425
|
+
fw="$(oci compute image get --image-id "$custom" --profile "$PROFILE" \
|
|
426
|
+
--query 'data."launch-options"."firmware"' --raw-output 2>/dev/null || true)"
|
|
427
|
+
if [[ "$fw" == "UEFI_64" ]]; then
|
|
428
|
+
image="$custom" mode="native"
|
|
429
|
+
else
|
|
430
|
+
say "note: newest esellar-alpine* custom image is firmware=${fw:-unknown} — A1 rejects BIOS-pinned imports, skipping to the platform-image + injection route"
|
|
431
|
+
fi
|
|
432
|
+
fi
|
|
433
|
+
if [[ -z "$image" ]]; then
|
|
434
|
+
image="$(oci compute instance get --instance-id "$oiid" --profile "$PROFILE" \
|
|
435
|
+
--query 'data."image-id"' --raw-output 2>/dev/null || true)"
|
|
436
|
+
[[ "$image" == ocid1.image* ]] || die "template instance has no resolvable image-id — cannot launch or inject"
|
|
437
|
+
mode="inject"
|
|
438
|
+
qcow2="${ALPINE_QCOW2:-${REPO_ROOT}/infra/alpine-host/build/esellar-alpine-3.22.6-aarch64.qcow2}"
|
|
439
|
+
[[ -f "$qcow2" ]] || die "golden qcow2 not found: $qcow2 (build via packer, RUNBOOK § 1, or set ALPINE_QCOW2)"
|
|
440
|
+
command -v qemu-img >/dev/null 2>&1 || die "qemu-img not found in PATH (brew install qemu) — required for the qcow2 -> raw conversion"
|
|
441
|
+
# ops ssh public key for the platform-image first boot. OPS_SSH_PUBKEY
|
|
442
|
+
# (explicit file) wins; else derive from the ssh AGENT (ssh-add -L) —
|
|
443
|
+
# the agent key is what every kampodine ssh + the golden image's baked
|
|
444
|
+
# ops key expect; ~/.ssh/id_ed25519.pub can be a stale personal key
|
|
445
|
+
# (2026-10-07 drill: launch with the stale pub = unreachable instance).
|
|
446
|
+
if [[ -n "${OPS_SSH_PUBKEY:-}" ]]; then
|
|
447
|
+
[[ -f "$OPS_SSH_PUBKEY" ]] || die "ops ssh public key not found: $OPS_SSH_PUBKEY (set OPS_SSH_PUBKEY or load the key into the agent)"
|
|
448
|
+
keyfile="$OPS_SSH_PUBKEY"
|
|
449
|
+
else
|
|
450
|
+
agent_keys="$(ssh-add -L 2>/dev/null | grep -v '\.pub$' || true)"
|
|
451
|
+
[[ -n "$agent_keys" ]] || die "no ssh key available: OPS_SSH_PUBKEY unset and ssh-add lists no keys (ssh-add ~/.ssh/id_ed25519-esellar, or set OPS_SSH_PUBKEY)"
|
|
452
|
+
keyfile="$(mktemp "${TMPDIR:-/tmp}/esellar-ops-pubkey.XXXXXX")"
|
|
453
|
+
printf '%s\n' "$agent_keys" > "$keyfile"
|
|
454
|
+
chmod 600 "$keyfile"
|
|
455
|
+
fi
|
|
456
|
+
fi
|
|
457
|
+
|
|
458
|
+
say "launching esellar-$color: image=${image:0:60}… ad=$oad subnet=${subnet:0:60}… mode=$mode"
|
|
459
|
+
# NOTE: --profile stays ON the launch line (script-gates static scan reads
|
|
460
|
+
# the invocation line, not array contents).
|
|
461
|
+
launch_args=(-c "$comp" --availability-domain "$oad" --subnet-id "$subnet"
|
|
462
|
+
--image-id "$image" --shape VM.Standard.A1.Flex --shape-config '{"ocpus":2,"memoryInGBs":12}'
|
|
463
|
+
--assign-public-ip true --display-name "esellar-$color")
|
|
464
|
+
[[ "$mode" == "inject" ]] && launch_args+=(--ssh-authorized-keys-file "$keyfile")
|
|
465
|
+
iid="$(oci compute instance launch "${launch_args[@]}" --profile "$PROFILE" --query 'data.id' --raw-output)"
|
|
156
466
|
say "LAUNCHED esellar-$color: $iid"
|
|
157
|
-
|
|
158
|
-
|
|
467
|
+
if [[ "$mode" == "native" ]]; then
|
|
468
|
+
say "next (RUNBOOK § start-fresh): wait RUNNING -> ssh in -> kampodine vm-prepare -> kampodine deploy --host root@<ephemeral-ip>"
|
|
469
|
+
say "then 'bluegreen.sh flip --to $color' (health-gated) once its app checks green."
|
|
470
|
+
exit 0
|
|
471
|
+
fi
|
|
472
|
+
say "waiting for RUNNING (inject mode)…"
|
|
473
|
+
instance_wait_running "$iid"
|
|
474
|
+
pubip="$(instance_by_color "$comp" "$color" | cut -d' ' -f2)"
|
|
475
|
+
[[ -n "$pubip" ]] || die "no ephemeral public ip on $iid yet — re-check with 'bluegreen status' and run the injection manually"
|
|
476
|
+
inject_alpine "$color" "$pubip" "$qcow2"
|
|
159
477
|
;;
|
|
160
478
|
|
|
161
479
|
flip)
|
|
@@ -168,53 +486,121 @@ case "$cmd" in
|
|
|
168
486
|
*) die "unknown flip flag: $1" ;;
|
|
169
487
|
esac
|
|
170
488
|
done
|
|
171
|
-
[[ "$to" == blue || "$to" == green ]] || { say "usage: kampodine bluegreen flip --to <blue|green> [--force]"; say "(rollback =
|
|
489
|
+
[[ "$to" == blue || "$to" == green ]] || { say "usage: kampodine bluegreen flip --to <blue|green> [--force]"; say "(rollback = unassign to dormant: 'kampodine bluegreen rollback')"; exit 2; }
|
|
172
490
|
comp="$(compartment_ocid)"
|
|
173
491
|
rp="$(reserved_ip "$comp")"
|
|
174
492
|
[[ -n "$rp" ]] || die "no reserved IP (run 'kampodine bluegreen init' first)"
|
|
175
493
|
read -r rocid raddr _ <<<"$rp"
|
|
494
|
+
rhost="${raddr//./-}.sslip.io"
|
|
176
495
|
trow="$(instance_by_color "$comp" "$to")"
|
|
177
496
|
[[ -n "$trow" ]] || die "esellar-$to is not RUNNING — nothing to flip to"
|
|
178
|
-
read -r tiid
|
|
179
|
-
|
|
180
|
-
|
|
497
|
+
read -r tiid tpub _ <<<"$trow"
|
|
498
|
+
tvnic="$(primary_vnic_of "$tiid")"
|
|
499
|
+
[[ -n "$tvnic" ]] || die "no primary VNIC on esellar-$to"
|
|
500
|
+
tpip="$(reserved_anchor_ip "$tvnic")"
|
|
501
|
+
[[ "$tpip" == ocid1.privateip* ]] || die "could not resolve/create the reserved-anchor secondary private ip on esellar-$to"
|
|
502
|
+
if instance_healthy "$tpub"; then
|
|
181
503
|
say "target health: esellar-$to app HEALTHY on its own IP"
|
|
182
504
|
else
|
|
183
505
|
(( force )) || die "esellar-$to app UNHEALTHY — refusing flip (override: --force)"
|
|
184
506
|
say "target health: UNHEALTHY — flipping anyway (--force)"
|
|
185
507
|
fi
|
|
186
|
-
|
|
508
|
+
|
|
509
|
+
# ACME-FIRST: HTTP-01 for the reserved-IP sslip name can only complete
|
|
510
|
+
# once the hostname resolves to the reserved ip AND routes to the target,
|
|
511
|
+
# so the guest anchor address goes FIRST, then the assign, then the cert.
|
|
512
|
+
ob=green; [[ "$to" == green ]] && ob=blue
|
|
513
|
+
say "flip[1/4]: anchor conf -> esellar-$to guest ($tpub), watcher configures the address"
|
|
514
|
+
tsubnet="$(oci compute vnic-attachment list -c "$comp" --profile "$PROFILE" \
|
|
515
|
+
--instance-id "$tiid" --query 'data[0]."subnet-id"' --raw-output 2>/dev/null || true)"
|
|
516
|
+
[[ "$tsubnet" == ocid1.subnet* ]] || die "could not resolve esellar-$to's subnet id"
|
|
517
|
+
tcidr="$(oci network subnet get --subnet-id "$tsubnet" --profile "$PROFILE" \
|
|
518
|
+
--query 'data."cidr-block"' --raw-output 2>/dev/null || true)"
|
|
519
|
+
[[ "$tcidr" =~ ^[0-9]+\.[0-9]+\.[0-9]+\.[0-9]+/[0-9]+$ ]] || die "could not resolve esellar-$to's subnet cidr (got: ${tcidr:-none})"
|
|
520
|
+
tanchor_addr="$(oci network private-ip get --private-ip-id "$tpip" --profile "$PROFILE" \
|
|
521
|
+
--query 'data."ip-address"' --raw-output 2>/dev/null || true)"
|
|
522
|
+
[[ "$tanchor_addr" =~ ^[0-9]+\.[0-9]+\.[0-9]+\.[0-9]+$ ]] || die "could not resolve the anchor private address on esellar-$to"
|
|
523
|
+
taddr_cidr="${tanchor_addr}/${tcidr##*/}"
|
|
524
|
+
if ! write_anchor_conf "$tpub" "$taddr_cidr" >/dev/null 2>&1; then
|
|
525
|
+
die "anchor.conf write failed on esellar-$to ($tpub) — nothing mutated, flip aborted"
|
|
526
|
+
fi
|
|
527
|
+
addr_ok=0
|
|
528
|
+
for ((i = 1; i <= FLIP_ADDR_TRIES; i++)); do
|
|
529
|
+
if anchor_addr_ready "$tpub" "$taddr_cidr" >/dev/null 2>&1; then addr_ok=1; break; fi
|
|
530
|
+
sleep "$FLIP_POLL_SLEEP"
|
|
531
|
+
done
|
|
532
|
+
if (( ! addr_ok )); then
|
|
533
|
+
cleanup_target_anchor "$tpub" "$taddr_cidr"
|
|
534
|
+
die "guest watcher never configured $taddr_cidr on esellar-$to — is rc-service esellar-anchor running? (conf removed, NOTHING mutated)"
|
|
535
|
+
fi
|
|
536
|
+
say "flip[2/4]: guest answers on $taddr_cidr — reserved $raddr -> esellar-$to (anchor $tpip)"
|
|
187
537
|
oci network public-ip update --public-ip-id "$rocid" --profile "$PROFILE" \
|
|
188
538
|
--private-ip-id "$tpip" --force --wait-for-state ASSIGNED >/dev/null \
|
|
189
|
-
|| die "OCI flip call failed —
|
|
190
|
-
|
|
191
|
-
if
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
say "public check FAILED through $raddr post-flip" >&2
|
|
195
|
-
ob=green; [[ "$to" == green ]] && ob=blue
|
|
196
|
-
say "auto-rollback: moving reserved IP back to esellar-$ob" >&2
|
|
197
|
-
brow="$(instance_by_color "$comp" "$ob")"
|
|
198
|
-
if [[ -n "$brow" ]]; then
|
|
199
|
-
bpip="$(primary_private_ip_ocid "$comp" "$(cut -d' ' -f1 <<<"$brow")")"
|
|
200
|
-
oci network public-ip update --public-ip-id "$rocid" --profile "$PROFILE" \
|
|
201
|
-
--private-ip-id "$bpip" --force >/dev/null \
|
|
202
|
-
&& say "rolled back to esellar-$ob — investigate esellar-$to before retrying" >&2 \
|
|
203
|
-
|| die "ROLLBACK FAILED — flip manually via console: $rocid"
|
|
204
|
-
fi
|
|
539
|
+
|| { cleanup_target_anchor "$tpub" "$taddr_cidr"; die "OCI flip call failed — guest conf removed; run 'kampodine bluegreen status'"; }
|
|
540
|
+
say "flip[3/4]: reserved IP ASSIGNED — registering $rhost on esellar-$to's kamal-proxy (ACME HTTP-01 through the reserved ip)…"
|
|
541
|
+
if ! ssh "${SSH_OPTS[@]}" "root@$tpub" \
|
|
542
|
+
"podman exec kamal-proxy kamal-proxy deploy esellar-api --host=$rhost --target=esellar-api:8080 --tls --health-check-path=/api/auth/ok"; then
|
|
543
|
+
flip_failure_rollback "$to" "$ob" "$comp" "$rocid" "$tpub" "$taddr_cidr"
|
|
205
544
|
exit 1
|
|
206
545
|
fi
|
|
546
|
+
acme_ok=0
|
|
547
|
+
for ((i = 1; i <= FLIP_ACME_TRIES; i++)); do
|
|
548
|
+
if curl -sf -m 8 --resolve "$rhost:443:$raddr" "https://$rhost/up"; then acme_ok=1; break; fi
|
|
549
|
+
sleep "$FLIP_POLL_SLEEP"
|
|
550
|
+
done
|
|
551
|
+
if (( ! acme_ok )); then
|
|
552
|
+
say "https://$rhost/ never answered with a VALID cert through $raddr (rate limits? HTTP-01 unreachable?)" >&2
|
|
553
|
+
flip_failure_rollback "$to" "$ob" "$comp" "$rocid" "$tpub" "$taddr_cidr"
|
|
554
|
+
exit 1
|
|
555
|
+
fi
|
|
556
|
+
served="$(curl -s -m 8 --resolve "$rhost:443:$raddr" "https://$rhost/api/auth/ok" || true)"
|
|
557
|
+
say "flip[4/4]: cert for $rhost VALID + serving through $raddr"
|
|
558
|
+
say "FLIPPED: https://$raddr/ (https://$rhost/) now serves from esellar-$to"
|
|
559
|
+
say "served /api/auth/ok: ${served:-<no body>}"
|
|
207
560
|
;;
|
|
208
561
|
|
|
209
562
|
rollback)
|
|
563
|
+
# DORMANT rollback (2026-10-07 pt3 contract): holder guest cleanup
|
|
564
|
+
# (anchor.conf removal + anchor address delete — the flip tool's explicit
|
|
565
|
+
# job; the esellar-anchor watcher is add-only) then the OCI unassign
|
|
566
|
+
# (documented CLI semantics: an empty --private-ip-id unassigns). For a
|
|
567
|
+
# real pair where traffic must land on the other color, use
|
|
568
|
+
# 'flip --to <other>' — it runs the full ACME-first sequence there.
|
|
210
569
|
comp="$(compartment_ocid)"
|
|
211
570
|
rp="$(reserved_ip "$comp")"
|
|
212
571
|
[[ -n "$rp" ]] || die "no reserved IP"
|
|
213
|
-
read -r
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
572
|
+
read -r rocid raddr rholder <<<"$rp"
|
|
573
|
+
if [[ -z "$rholder" || "$rholder" == "-" ]]; then
|
|
574
|
+
say "reserved $raddr already UNASSIGNED (dormant) — nothing to roll back"
|
|
575
|
+
exit 0
|
|
576
|
+
fi
|
|
577
|
+
# resolve the holder color by matching its anchor (LOOKUP-ONLY)
|
|
578
|
+
holder_color="" holder_ip=""
|
|
579
|
+
for color in blue green; do
|
|
580
|
+
crow="$(instance_by_color "$comp" "$color")"
|
|
581
|
+
[[ -n "$crow" ]] || continue
|
|
582
|
+
cnic="$(primary_vnic_of "$(cut -d' ' -f1 <<<"$crow")")"
|
|
583
|
+
capip=""
|
|
584
|
+
[[ -n "$cnic" ]] && capip="$(existing_anchor_ip "$cnic")"
|
|
585
|
+
if [[ "$capip" == "$rholder" ]]; then
|
|
586
|
+
holder_color="$color"
|
|
587
|
+
holder_ip="$(cut -d' ' -f2 <<<"$crow")"
|
|
588
|
+
break
|
|
589
|
+
fi
|
|
590
|
+
done
|
|
591
|
+
if [[ -z "$holder_color" ]]; then
|
|
592
|
+
say "warning: reserved $raddr held by an anchor of neither RUNNING color (terminated instance?) — unassigning without guest cleanup"
|
|
593
|
+
else
|
|
594
|
+
say "current holder: esellar-$holder_color ($holder_ip) — cleaning the guest anchor, then unassigning"
|
|
595
|
+
conf_addr="$(read_anchor_conf_addr "$holder_ip")"
|
|
596
|
+
cleanup_target_anchor "$holder_ip" "$conf_addr"
|
|
597
|
+
say "esellar-$holder_color guest cleaned (anchor.conf removed${conf_addr:+, address $conf_addr deleted})"
|
|
598
|
+
fi
|
|
599
|
+
say "unassigning reserved $raddr (-> dormant)…"
|
|
600
|
+
oci network public-ip update --public-ip-id "$rocid" --profile "$PROFILE" \
|
|
601
|
+
--private-ip-id "" --force --wait-for-state AVAILABLE >/dev/null \
|
|
602
|
+
|| die "unassign failed — check 'kampodine bluegreen status' and the console"
|
|
603
|
+
say "ROLLED BACK: reserved $raddr UNASSIGNED (dormant)"
|
|
218
604
|
;;
|
|
219
605
|
|
|
220
606
|
*)
|
package/scripts/image-import.sh
CHANGED
|
@@ -14,6 +14,8 @@
|
|
|
14
14
|
# Auth: ~/.oci/config (brew install oci-cli; oci setup config) by default, or
|
|
15
15
|
# --from-pass to source API credentials from the local pass store — the key
|
|
16
16
|
# NEVER enters this repo, the image, or any log (values only go into env vars).
|
|
17
|
+
# Every oci call pins --profile $PROFILE (OCI_PROFILE, default esellar-api) —
|
|
18
|
+
# the CLI's DEFAULT profile is NOT the esellar tenancy.
|
|
17
19
|
# Placeholder pass paths (create yours to match):
|
|
18
20
|
# esellar/oci/user user OCID
|
|
19
21
|
# esellar/oci/tenancy tenancy OCID
|
|
@@ -31,6 +33,7 @@ IMAGE=""
|
|
|
31
33
|
BUCKET="${OCI_IMPORT_BUCKET:-esellar-image-import}"
|
|
32
34
|
NAME_PREFIX="${OCI_IMPORT_NAME:-esellar-alpine}"
|
|
33
35
|
COMPARTMENT_NAME="${OCI_COMPARTMENT:-esellar}"
|
|
36
|
+
PROFILE="${OCI_PROFILE:-esellar-api}"
|
|
34
37
|
FROM_PASS=0
|
|
35
38
|
KEEP_OBJECT=0
|
|
36
39
|
|
|
@@ -76,9 +79,9 @@ if [[ -z "$IMAGE" ]]; then
|
|
|
76
79
|
fi
|
|
77
80
|
[[ -f "$IMAGE" ]] || die "image not found: $IMAGE"
|
|
78
81
|
|
|
79
|
-
COMPARTMENT_OCID="$(oci iam compartment list --all --query "data[?name=='$COMPARTMENT_NAME'].id | [0]" --raw-output 2>/dev/null || true)"
|
|
80
|
-
[[ "$COMPARTMENT_OCID" == ocid1.compartment* ]] || die "compartment '$COMPARTMENT_NAME' not found (
|
|
81
|
-
NAMESPACE="$(oci os ns get --query data --raw-output)"
|
|
82
|
+
COMPARTMENT_OCID="$(oci iam compartment list --all --profile "$PROFILE" --query "data[?name=='$COMPARTMENT_NAME'].id | [0]" --raw-output 2>/dev/null || true)"
|
|
83
|
+
[[ "$COMPARTMENT_OCID" == ocid1.compartment* ]] || die "compartment '$COMPARTMENT_NAME' not found (profile $PROFILE)"
|
|
84
|
+
NAMESPACE="$(oci os ns get --profile "$PROFILE" --query data --raw-output)"
|
|
82
85
|
[[ -n "$NAMESPACE" ]] || die "could not resolve the tenancy namespace"
|
|
83
86
|
|
|
84
87
|
STAMP="$(date +%Y%m%d-%H%M%S)"
|
|
@@ -86,11 +89,23 @@ OBJECT_NAME="${NAME_PREFIX}-${STAMP}.qcow2"
|
|
|
86
89
|
IMAGE_NAME="${NAME_PREFIX}-$(basename "$IMAGE" .qcow2 | sed 's/^esellar-alpine-//')-${STAMP}"
|
|
87
90
|
|
|
88
91
|
say "uploading $(basename "$IMAGE") -> os://$BUCKET/$OBJECT_NAME"
|
|
89
|
-
oci os object put -bn "$BUCKET" --file "$IMAGE" --name "$OBJECT_NAME" --force \
|
|
92
|
+
oci os object put -bn "$BUCKET" --profile "$PROFILE" --file "$IMAGE" --name "$OBJECT_NAME" --force \
|
|
90
93
|
|| die "object upload failed (bucket exists? oci os bucket create -bn $BUCKET -c $COMPARTMENT_OCID)"
|
|
91
94
|
|
|
92
|
-
say "importing as custom image '$IMAGE_NAME' (self-supported:
|
|
93
|
-
|
|
95
|
+
say "importing as custom image '$IMAGE_NAME' (self-supported: PARAVIRTUALIZED)…"
|
|
96
|
+
# A1/Ampere firmware gate (2026-10-07 drill, empirically exhausted):
|
|
97
|
+
# OCI pins imported images to launch-options firmware=BIOS. It is NOT
|
|
98
|
+
# derived from the --operating-system string (a recognized aarch64 string
|
|
99
|
+
# still yields BIOS) and NOT re-derivable via `compute image update` (OS
|
|
100
|
+
# metadata is writable, firmware is frozen). A1 shape launch validation
|
|
101
|
+
# rejects BIOS images — "Shape VM.Standard.A1.Flex is not valid for image"
|
|
102
|
+
# — and NO launch-time override passes it: --launch-options firmware /
|
|
103
|
+
# --launch-mode CUSTOM were both rejected. There is also no image->boot-
|
|
104
|
+
# volume API (bootVolumes sourceDetails rejects type "image"). The only
|
|
105
|
+
# sanctioned route to an A1-launchable custom image is CAPTURE FROM A
|
|
106
|
+
# RUNNING A1 INSTANCE (compute image create --instance-id). The check
|
|
107
|
+
# below warns loudly when the import lands BIOS.
|
|
108
|
+
IMPORT_JSON="$(oci compute image import from-object --profile "$PROFILE" \
|
|
94
109
|
-c "$COMPARTMENT_OCID" \
|
|
95
110
|
--bucket-name "$BUCKET" \
|
|
96
111
|
--namespace "$NAMESPACE" \
|
|
@@ -98,7 +113,7 @@ IMPORT_JSON="$(oci compute image import from-object \
|
|
|
98
113
|
--display-name "$IMAGE_NAME" \
|
|
99
114
|
--source-image-type QCOW2 \
|
|
100
115
|
--operating-system "Linux" \
|
|
101
|
-
--operating-system-version "Alpine (self-supported)" \
|
|
116
|
+
--operating-system-version "Alpine 3.22.6 (self-supported)" \
|
|
102
117
|
--launch-mode PARAVIRTUALIZED \
|
|
103
118
|
--query 'data.id' --raw-output)" || die "import call failed"
|
|
104
119
|
[[ "$IMPORT_JSON" == ocid1.image* ]] || die "unexpected import response: $IMPORT_JSON"
|
|
@@ -106,7 +121,7 @@ IMPORT_JSON="$(oci compute image import from-object \
|
|
|
106
121
|
say "polling import -> AVAILABLE (up to 30m)…"
|
|
107
122
|
STATE="PENDING_IMPORT"
|
|
108
123
|
for _ in $(seq 1 120); do
|
|
109
|
-
STATE="$(oci compute image get --image-id "$IMPORT_JSON" --query 'data."lifecycle-state"' --raw-output 2>/dev/null || echo UNKNOWN)"
|
|
124
|
+
STATE="$(oci compute image get --image-id "$IMPORT_JSON" --profile "$PROFILE" --query 'data."lifecycle-state"' --raw-output 2>/dev/null || echo UNKNOWN)"
|
|
110
125
|
case "$STATE" in
|
|
111
126
|
AVAILABLE) break ;;
|
|
112
127
|
PENDING_IMPORT|IMPORTING|UPLOADING) sleep 15 ;;
|
|
@@ -115,13 +130,29 @@ for _ in $(seq 1 120); do
|
|
|
115
130
|
done
|
|
116
131
|
[[ "$STATE" == "AVAILABLE" ]] || die "import did not become AVAILABLE in 30m (state: $STATE)"
|
|
117
132
|
|
|
133
|
+
# Firmware verdict — A1/Ampere (our target shape) rejects BIOS-pinned images
|
|
134
|
+
# at launch (see the import-call comment above). Loud, non-fatal: other
|
|
135
|
+
# (x86) shapes launch fine from a BIOS image.
|
|
136
|
+
IMPORT_FIRMWARE="$(oci compute image get --image-id "$IMPORT_JSON" --profile "$PROFILE" --query 'data."launch-options"."firmware"' --raw-output 2>/dev/null || echo UNKNOWN)"
|
|
137
|
+
if [[ "$IMPORT_FIRMWARE" != "UEFI_64" ]]; then
|
|
138
|
+
say "WARN: import landed firmware=$IMPORT_FIRMWARE — VM.Standard.A1.Flex (Ampere, UEFI-only) WILL reject this image at launch."
|
|
139
|
+
say " OCI has no import-time firmware control (drill-proven 2026-10-07): the sanctioned A1 route is"
|
|
140
|
+
say " capture-from-instance: boot the qcow2 elsewhere, then 'oci compute image create --instance-id <running-a1>'."
|
|
141
|
+
say " x86 shapes CAN launch this image directly. See infra/alpine-host/RUNBOOK.md § blue-green."
|
|
142
|
+
fi
|
|
143
|
+
|
|
118
144
|
if [[ $KEEP_OBJECT -eq 0 ]]; then
|
|
119
145
|
say "deleting the staged object (image data now lives in the custom image)…"
|
|
120
|
-
oci os object delete -bn "$BUCKET" --name "$OBJECT_NAME" --force >/dev/null || say "WARN: staged object delete failed (cleanup manually)"
|
|
146
|
+
oci os object delete -bn "$BUCKET" --profile "$PROFILE" --name "$OBJECT_NAME" --force >/dev/null || say "WARN: staged object delete failed (cleanup manually)"
|
|
121
147
|
fi
|
|
122
148
|
|
|
123
149
|
say "IMPORTED:"
|
|
124
150
|
say " image OCID: $IMPORT_JSON"
|
|
125
151
|
say " display : $IMAGE_NAME"
|
|
152
|
+
say " firmware : $IMPORT_FIRMWARE"
|
|
126
153
|
say "next:"
|
|
127
|
-
|
|
154
|
+
if [[ "$IMPORT_FIRMWARE" == "UEFI_64" ]]; then
|
|
155
|
+
say " A1-ready. kampodine bluegreen provision <blue|green> picks this image (newest esellar-alpine*)."
|
|
156
|
+
else
|
|
157
|
+
say " A1 launch will REJECT this image (firmware $IMPORT_FIRMWARE) — see the WARN above / RUNBOOK § blue-green."
|
|
158
|
+
fi
|
package/scripts/migrate.sh
CHANGED
|
@@ -55,17 +55,31 @@ fi
|
|
|
55
55
|
failures=0
|
|
56
56
|
migrated=0
|
|
57
57
|
|
|
58
|
+
RESULTS_DIR="$(mktemp -d /tmp/migrate-results.XXXXXX)"
|
|
59
|
+
trap 'rm -rf "$RESULTS_DIR"' EXIT
|
|
60
|
+
|
|
58
61
|
migrate_one() {
|
|
59
62
|
local db_file="$1" ns="$2"
|
|
63
|
+
local marker
|
|
64
|
+
marker="${RESULTS_DIR}/$(basename "$db_file" .db).status"
|
|
60
65
|
log "migrating ${db_file#"${TENANT_DIR}"/} (ns: ${ns})"
|
|
61
66
|
if (cd "$REPO_ROOT" && pnpm --filter api exec tsx scripts/libsql-migrate/migrate-db.ts --db "file:${db_file}" --ns "$ns"); then
|
|
62
|
-
|
|
67
|
+
printf 'ok' > "$marker"
|
|
63
68
|
else
|
|
69
|
+
printf 'FAIL' > "$marker"
|
|
64
70
|
printf '[migrate-all][FAIL] migration failed for %s\n' "$db_file" >&2
|
|
65
|
-
failures=$((failures + 1))
|
|
66
71
|
fi
|
|
67
72
|
}
|
|
68
73
|
|
|
74
|
+
tally() {
|
|
75
|
+
failures=0; migrated=0
|
|
76
|
+
local m
|
|
77
|
+
for m in "$RESULTS_DIR"/*.status; do
|
|
78
|
+
[[ -e "$m" ]] || continue
|
|
79
|
+
if [[ "$(cat "$m")" = "ok" ]]; then migrated=$((migrated + 1)); else failures=$((failures + 1)); fi
|
|
80
|
+
done
|
|
81
|
+
}
|
|
82
|
+
|
|
69
83
|
# root.db (auth/org plane) migrates FIRST. App seam (sqld-topology.ts rootDbPath)
|
|
70
84
|
# resolves it at /data/tenants/root/root.db; a legacy flat /data/tenants/root.db
|
|
71
85
|
# is honored as fallback.
|
|
@@ -75,16 +89,28 @@ elif [[ -f "${TENANT_DIR}/root.db" ]]; then
|
|
|
75
89
|
migrate_one "${TENANT_DIR}/root.db" root
|
|
76
90
|
fi
|
|
77
91
|
|
|
78
|
-
# then tenant files sorted
|
|
92
|
+
# then tenant files sorted — bounded-parallel by default (-j 4): per-file
|
|
93
|
+
# write locks are independent (one writer per file by the store model), so
|
|
94
|
+
# parallelism is safe; the failure gate stays all-or-nothing either way.
|
|
95
|
+
JOBS="${MIGRATE_JOBS:-4}"
|
|
79
96
|
shopt -s nullglob
|
|
80
97
|
tenant_files=("$TENANT_DIR"/tenant_*.db)
|
|
81
98
|
shopt -u nullglob
|
|
82
99
|
if [[ ${#tenant_files[@]} -gt 0 ]]; then
|
|
83
100
|
sorted_tenants=()
|
|
84
101
|
while IFS= read -r f; do sorted_tenants+=("$f"); done < <(printf '%s\n' "${tenant_files[@]}" | sort)
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
102
|
+
if [[ "$JOBS" -le 1 ]]; then
|
|
103
|
+
for db_file in "${sorted_tenants[@]}"; do
|
|
104
|
+
migrate_one "$db_file" "$(basename "$db_file" .db)"
|
|
105
|
+
done
|
|
106
|
+
else
|
|
107
|
+
# migrate_one is marker-file based (no shared shell state) — xargs
|
|
108
|
+
# children report through $RESULTS_DIR; tally() aggregates after.
|
|
109
|
+
export RESULTS_DIR TENANT_DIR REPO_ROOT
|
|
110
|
+
export -f migrate_one log
|
|
111
|
+
printf '%s\n' "${sorted_tenants[@]}" | xargs -P "$JOBS" -I{} bash -c 'migrate_one "$@" "$(basename "$1" .db)"' _ {}
|
|
112
|
+
fi
|
|
113
|
+
tally
|
|
88
114
|
fi
|
|
89
115
|
|
|
90
116
|
if [[ $failures -gt 0 ]]; then
|
package/scripts/vm-prepare.sh
CHANGED
|
@@ -8,7 +8,9 @@
|
|
|
8
8
|
#
|
|
9
9
|
# Alpine ships NO systemd anywhere (verified v3.22 main/community + edge; the
|
|
10
10
|
# golden image boots OpenRC) — everything here is OpenRC-native:
|
|
11
|
-
# 1. sanity gates (UEFI boot, OpenRC tooling
|
|
11
|
+
# 1. sanity gates (UEFI boot, OpenRC tooling; sshd hardening is
|
|
12
|
+
# ENSURED — drop-in + Include + restart — then gated: the golden image
|
|
13
|
+
# predates the baked hardening, RUNBOOK §Blue-green pt2 gap)
|
|
12
14
|
# 2. apk repositories: ensure the v3.22 community repo (podman lives there)
|
|
13
15
|
# 3. podman stack: podman podman-docker crun catatonit netavark
|
|
14
16
|
# aardvark-dns fuse-overlayfs
|
|
@@ -16,9 +18,13 @@
|
|
|
16
18
|
# 5. /etc/containers/registries.conf (insecure 127.0.0.1:5000; search docker.io)
|
|
17
19
|
# 6. sysctl net.ipv4.ip_unprivileged_port_start=80 (persisted + applied live)
|
|
18
20
|
# 7. OpenRC services /etc/init.d/esellar-api + /etc/init.d/kamal-proxy
|
|
19
|
-
# (supervise-daemon around plain `podman run`),
|
|
21
|
+
# (supervise-daemon around plain `podman run`), rc-update'd into the
|
|
20
22
|
# default runlevel — that IS boot survival (no quadlets, no systemctl)
|
|
21
|
-
# 8.
|
|
23
|
+
# 8. esellar-anchor: the blue-green reserved-ip flip's GUEST half — a
|
|
24
|
+
# busybox watcher (no container) that polls /etc/esellar/anchor.conf and
|
|
25
|
+
# `ip addr add`s the anchor address at flip time. Started+enabled on every
|
|
26
|
+
# VM, inert without the conf (flip tooling writes it over ssh).
|
|
27
|
+
# 9. kamal-proxy image pulled + service UP — the first deploy execs into it
|
|
22
28
|
# to issue the fresh ACME certificate
|
|
23
29
|
#
|
|
24
30
|
# The esellar-api container is NOT started here: neither its image (pulled by
|
|
@@ -194,6 +200,137 @@ stop_post() {
|
|
|
194
200
|
}
|
|
195
201
|
INITD_PROXY
|
|
196
202
|
|
|
203
|
+
cat > "$TMPDIR_LOCAL/esellar-anchor.sh" <<'ANCHOR_WATCHER'
|
|
204
|
+
#!/bin/sh
|
|
205
|
+
# KEEP IN SYNC with ansible/roles/container-service/files/esellar-anchor.sh
|
|
206
|
+
# (vm-prepare rendered-content convention: vm-prepare bootstraps a fresh VM
|
|
207
|
+
# BEFORE the first ansible converge; ansible owns the file afterwards).
|
|
208
|
+
#
|
|
209
|
+
# esellar-anchor.sh — guest half of the blue-green reserved-ip flip.
|
|
210
|
+
#
|
|
211
|
+
# OCI assigns the reserved PUBLIC ip to a SECONDARY private ip ("the anchor")
|
|
212
|
+
# on the instance VNIC, but this image has NO oracle-cloud-agent: nothing
|
|
213
|
+
# configures that private ip inside the guest, so packets to the reserved ip
|
|
214
|
+
# die until the address exists on the interface (RUNBOOK §Blue-green,
|
|
215
|
+
# 2026-10-07 pt2 serving-leg gap). `kampodine bluegreen flip` writes
|
|
216
|
+
# /etc/esellar/anchor.conf over ssh at flip time; this watcher polls it and
|
|
217
|
+
# runs `ip addr add` within one interval. The unit is enabled+started on every
|
|
218
|
+
# VM by default and is INERT without the conf — a VM that never flips never
|
|
219
|
+
# touches its addresses.
|
|
220
|
+
#
|
|
221
|
+
# CONF (written by the flip tooling, root-only dir):
|
|
222
|
+
# ANCHOR_ADDR="10.0.0.14/24" # anchor private ip + subnet prefix (required)
|
|
223
|
+
# ANCHOR_IFACE="eth0" # optional; empty/unset = default-route iface
|
|
224
|
+
#
|
|
225
|
+
# ADD-ONLY BY DESIGN: this script NEVER removes or flushes addresses (rollback
|
|
226
|
+
# cleanup is the flip tool's explicit ssh job). Idempotent: a present address
|
|
227
|
+
# is a no-op. Only state TRANSITIONS are logged (bounded log noise), to
|
|
228
|
+
# stdout — supervise-daemon tees it to /var/log/esellar-anchor.log.
|
|
229
|
+
|
|
230
|
+
CONF="/etc/esellar/anchor.conf"
|
|
231
|
+
INTERVAL="${ESPELLAR_ANCHOR_INTERVAL:-5}"
|
|
232
|
+
STATE="boot" # last logged transition (boot | idle | added | error)
|
|
233
|
+
|
|
234
|
+
log() { printf '%s esellar-anchor: %s\n' "$(date '+%Y-%m-%dT%H:%M:%S%z')" "$*"; }
|
|
235
|
+
|
|
236
|
+
# supervise-daemon SIGTERMs on stop — die cleanly with the current tick.
|
|
237
|
+
trap 'exit 0' TERM INT
|
|
238
|
+
|
|
239
|
+
default_iface() {
|
|
240
|
+
ip -4 route show default 2>/dev/null | awk '{print $5; exit}'
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
# Dotted-quad/prefix sanity gate before anything touches `ip addr add` — the
|
|
244
|
+
# conf is operator-tooling written, but sourcing it must never turn garbage
|
|
245
|
+
# into an `ip` invocation.
|
|
246
|
+
addr_wellformed() {
|
|
247
|
+
case "$1" in
|
|
248
|
+
[0-9]*.[0-9]*.[0-9]*.[0-9]*/*) return 0 ;;
|
|
249
|
+
*) return 1 ;;
|
|
250
|
+
esac
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
while :; do
|
|
254
|
+
ANCHOR_ADDR=""
|
|
255
|
+
ANCHOR_IFACE=""
|
|
256
|
+
# shellcheck disable=SC1090
|
|
257
|
+
[ -f "$CONF" ] && . "$CONF"
|
|
258
|
+
if [ -n "${ANCHOR_ADDR:-}" ] && addr_wellformed "$ANCHOR_ADDR"; then
|
|
259
|
+
iface="${ANCHOR_IFACE:-$(default_iface)}"
|
|
260
|
+
if [ -n "$iface" ] && ip -4 addr show dev "$iface" 2>/dev/null | grep -qF -- "$ANCHOR_ADDR"; then
|
|
261
|
+
[ "$STATE" = added ] || { log "anchor $ANCHOR_ADDR present on $iface"; STATE=added; }
|
|
262
|
+
elif [ -n "$iface" ] && ip addr add "$ANCHOR_ADDR" dev "$iface" 2>/dev/null; then
|
|
263
|
+
log "anchor ADDED $ANCHOR_ADDR dev $iface"
|
|
264
|
+
STATE=added
|
|
265
|
+
else
|
|
266
|
+
[ "$STATE" = error ] || { log "anchor add FAILED for $ANCHOR_ADDR (iface ${iface:-none})"; STATE=error; }
|
|
267
|
+
fi
|
|
268
|
+
else
|
|
269
|
+
[ "$STATE" = idle ] || { log "no anchor conf (${CONF}) — idling"; STATE=idle; }
|
|
270
|
+
fi
|
|
271
|
+
sleep "$INTERVAL"
|
|
272
|
+
done
|
|
273
|
+
ANCHOR_WATCHER
|
|
274
|
+
|
|
275
|
+
cat > "$TMPDIR_LOCAL/esellar-anchor" <<'INITD_ANCHOR'
|
|
276
|
+
#!/sbin/openrc-run
|
|
277
|
+
# Managed by packages/kampodine/scripts/vm-prepare.sh — do not hand-edit.
|
|
278
|
+
# KEEP IN SYNC with ansible/roles/container-service/templates/esellar-anchor.initd.j2
|
|
279
|
+
# (rendered with role defaults — see the esellar-api header for the
|
|
280
|
+
# vm-prepare/ansible split).
|
|
281
|
+
#
|
|
282
|
+
# Guest half of the blue-green reserved-ip flip: supervise-daemon runs the
|
|
283
|
+
# esellar-anchor.sh watcher, which polls /etc/esellar/anchor.conf (written by
|
|
284
|
+
# `kampodine bluegreen flip` over ssh at flip time) and `ip addr add`s the
|
|
285
|
+
# anchor address when it appears. Started+enabled on EVERY vm by default and
|
|
286
|
+
# INERT without the conf: a VM that never flips never touches its addresses —
|
|
287
|
+
# unlike the container units, there is no esellar_start_containers gate.
|
|
288
|
+
# Transitions land in /var/log/esellar-anchor.log.
|
|
289
|
+
|
|
290
|
+
name="esellar-anchor"
|
|
291
|
+
description="Blue-green reserved-ip anchor address watcher (guest half of the flip)"
|
|
292
|
+
|
|
293
|
+
supervisor=supervise-daemon
|
|
294
|
+
command="/usr/local/sbin/esellar-anchor.sh"
|
|
295
|
+
|
|
296
|
+
# Respawn forever: the watcher itself never exits (trap TERM/INT -> exit 0 on
|
|
297
|
+
# stop), so a respawn means the script died abnormally — retry gently.
|
|
298
|
+
respawn_delay=5
|
|
299
|
+
respawn_max=0
|
|
300
|
+
|
|
301
|
+
supervise_daemon_args="--stdout /var/log/esellar-anchor.log --stderr /var/log/esellar-anchor.log"
|
|
302
|
+
|
|
303
|
+
depend() {
|
|
304
|
+
need net
|
|
305
|
+
}
|
|
306
|
+
INITD_ANCHOR
|
|
307
|
+
|
|
308
|
+
cat > "$TMPDIR_LOCAL/sshd-hardening.conf" <<'SSHD_HARDENING'
|
|
309
|
+
# KEEP IN SYNC with ansible/roles/alpine-base/files/sshd-hardening.conf
|
|
310
|
+
# (byte-for-byte — ansible owns the file afterwards, rendered-content
|
|
311
|
+
# convention). Ensured BEFORE the hardening gate below: the golden image
|
|
312
|
+
# predates the baked drop-in (RUNBOOK §Blue-green pt2 gap, closed pt3) and
|
|
313
|
+
# `sshd -T` reports passwordauthentication yes on a fresh VM.
|
|
314
|
+
# Keys-only management: this VM's ssh surface is the ONLY management path
|
|
315
|
+
# (OCI security lists keep 22 closed to the world; access via temporary
|
|
316
|
+
# scoped rule or bastion — infra/oci/README.md model still applies).
|
|
317
|
+
PasswordAuthentication no
|
|
318
|
+
KbdInteractiveAuthentication no
|
|
319
|
+
PermitRootLogin prohibit-password
|
|
320
|
+
PermitEmptyPasswords no
|
|
321
|
+
MaxAuthTries 3
|
|
322
|
+
X11Forwarding no
|
|
323
|
+
# Reverse tunnel support for the Mac-local registry pull path
|
|
324
|
+
# (ssh -R 5000:... from kampodine deploy) rides the default AllowTcpForwarding.
|
|
325
|
+
GatewayPorts no
|
|
326
|
+
ClientAliveInterval 30
|
|
327
|
+
ClientAliveCountMax 6
|
|
328
|
+
|
|
329
|
+
# deploy registry tunnel rides a REMOTE forward (ssh -R 5000) — Alpine stock
|
|
330
|
+
# sshd_config ships AllowTcpForwarding no; remote-only keeps -L clients off.
|
|
331
|
+
AllowTcpForwarding yes
|
|
332
|
+
SSHD_HARDENING
|
|
333
|
+
|
|
197
334
|
# --- 0. connectivity ----------------------------------------------------------
|
|
198
335
|
say "waiting for ssh on ${HOST}…"
|
|
199
336
|
READY=0
|
|
@@ -213,8 +350,16 @@ vm '! readlink /proc/1/exe 2>/dev/null | grep -q systemd' || die "systemd is PID
|
|
|
213
350
|
vm 'command -v openrc >/dev/null && command -v rc-service >/dev/null && command -v rc-update >/dev/null && command -v supervise-daemon >/dev/null && rc-status --servicelist >/dev/null' \
|
|
214
351
|
|| die "OpenRC tooling missing or not operational (openrc/rc-service/rc-update/supervise-daemon, rc-status)"
|
|
215
352
|
|
|
353
|
+
say "sshd hardening: ensure drop-in + Include + restart (golden image predates the baked hardening — then the gate below holds)…"
|
|
354
|
+
# shellcheck disable=SC2016
|
|
355
|
+
vm 'grep -q "^Include /etc/ssh/sshd_config.d/\*.conf" /etc/ssh/sshd_config || sed -i "1i Include /etc/ssh/sshd_config.d/*.conf" /etc/ssh/sshd_config' \
|
|
356
|
+
|| die "could not ensure the sshd_config Include line"
|
|
357
|
+
vm 'mkdir -p /etc/ssh/sshd_config.d && chmod 700 /etc/ssh/sshd_config.d'
|
|
358
|
+
scp -q "${SSH_ARGS[@]}" "$TMPDIR_LOCAL/sshd-hardening.conf" "$HOST:/etc/ssh/sshd_config.d/99-esellar-hardening.conf"
|
|
359
|
+
vm 'rc-service sshd restart' || die "sshd restart failed after hardening ensure"
|
|
360
|
+
|
|
216
361
|
say "gate: sshd hardening effective"
|
|
217
|
-
vm 'sshd -T 2>/dev/null | grep -qi "^passwordauthentication no"' || die "sshd still allows passwords (
|
|
362
|
+
vm 'sshd -T 2>/dev/null | grep -qi "^passwordauthentication no"' || die "sshd still allows passwords (hardening ensure failed?)"
|
|
218
363
|
|
|
219
364
|
# --- 2. apk repositories + podman stack ----------------------------------------
|
|
220
365
|
say "ensuring the v3.22 community repo (podman stack lives there)…"
|
|
@@ -272,6 +417,25 @@ say "installing OpenRC services esellar-api + kamal-proxy (supervise-daemon arou
|
|
|
272
417
|
scp -q "${SSH_ARGS[@]}" "$TMPDIR_LOCAL/esellar-api" "$HOST:/etc/init.d/esellar-api"
|
|
273
418
|
scp -q "${SSH_ARGS[@]}" "$TMPDIR_LOCAL/kamal-proxy" "$HOST:/etc/init.d/kamal-proxy"
|
|
274
419
|
vm 'chmod 755 /etc/init.d/esellar-api /etc/init.d/kamal-proxy'
|
|
420
|
+
|
|
421
|
+
# --- 4b. esellar-anchor (blue-green flip guest half — watcher, no container) ----
|
|
422
|
+
# Ships on EVERY vm: started + enabled now, INERT until a flip writes
|
|
423
|
+
# /etc/esellar/anchor.conf over ssh (ansible container-service owns both files
|
|
424
|
+
# afterwards — drift repair keeps them in sync).
|
|
425
|
+
say "installing the esellar-anchor watcher service (blue-green flip guest half)…"
|
|
426
|
+
# /usr/local/sbin does NOT exist on the golden image (fresh Alpine ships no
|
|
427
|
+
# /usr/local hierarchy; the alpine-base role creates it — vm-prepare runs
|
|
428
|
+
# BEFORE any ansible)
|
|
429
|
+
vm 'mkdir -p /usr/local/sbin' || die "mkdir /usr/local/sbin failed"
|
|
430
|
+
scp -q "${SSH_ARGS[@]}" "$TMPDIR_LOCAL/esellar-anchor.sh" "$HOST:/usr/local/sbin/esellar-anchor.sh"
|
|
431
|
+
scp -q "${SSH_ARGS[@]}" "$TMPDIR_LOCAL/esellar-anchor" "$HOST:/etc/init.d/esellar-anchor"
|
|
432
|
+
vm 'chmod 755 /usr/local/sbin/esellar-anchor.sh /etc/init.d/esellar-anchor' || die "chmod esellar-anchor files failed"
|
|
433
|
+
vm 'sh -n /usr/local/sbin/esellar-anchor.sh' || die "esellar-anchor.sh does not parse (busybox sh)"
|
|
434
|
+
vm 'sh -n /etc/init.d/esellar-anchor' || die "/etc/init.d/esellar-anchor does not parse"
|
|
435
|
+
vm 'rc-update show default | grep -qE "^[[:space:]]*esellar-anchor[[:space:]]*\\|" || rc-update add esellar-anchor default' \
|
|
436
|
+
|| die "rc-update add esellar-anchor default failed"
|
|
437
|
+
vm 'rc-service esellar-anchor start' || die "rc-service esellar-anchor start failed"
|
|
438
|
+
|
|
275
439
|
for SVC in esellar-api kamal-proxy; do
|
|
276
440
|
vm "rc-update show default | grep -qE '^[[:space:]]*${SVC}[[:space:]]*\\|' || rc-update add ${SVC} default" \
|
|
277
441
|
|| die "rc-update add $SVC default failed"
|
|
@@ -281,7 +445,7 @@ done
|
|
|
281
445
|
# re-runs after a deploy must not bounce production.
|
|
282
446
|
# shellcheck disable=SC2016
|
|
283
447
|
vm_sh <<'UNIT_DRIFT' || true
|
|
284
|
-
for SVC in esellar-api kamal-proxy; do
|
|
448
|
+
for SVC in esellar-api kamal-proxy esellar-anchor; do
|
|
285
449
|
if rc-service "$SVC" status > /dev/null 2>&1; then
|
|
286
450
|
echo "RUNNING: $SVC (unit file was overwritten — rc-service $SVC restart to apply, on your call)"
|
|
287
451
|
fi
|
|
@@ -304,12 +468,14 @@ done
|
|
|
304
468
|
# --- 6. optional: pre-pull the app image over the registry tunnel ---------------
|
|
305
469
|
if [[ $DO_PULL -eq 1 ]]; then
|
|
306
470
|
say "pre-pulling the app image through the registry tunnel (fail BEFORE the first deploy)…"
|
|
307
|
-
|
|
471
|
+
# busybox wget, NOT curl: the golden image ships no curl (alpine-base
|
|
472
|
+
# installs it later; vm-prepare runs BEFORE any ansible — pt3 live)
|
|
473
|
+
vm 'busybox wget -q -O /dev/null http://127.0.0.1:5000/v2/' || {
|
|
308
474
|
pkill -f "ssh.*-R 5000" 2>/dev/null || true; sleep 1
|
|
309
475
|
nohup ssh -R 5000:127.0.0.1:5000 -N -o ServerAliveInterval=30 -o ExitOnForwardFailure=yes "$HOST" >/tmp/esellar-tunnel.log 2>&1 &
|
|
310
476
|
sleep 3
|
|
311
477
|
}
|
|
312
|
-
vm '
|
|
478
|
+
vm 'busybox wget -q -O /dev/null http://127.0.0.1:5000/v2/' || die "registry tunnel did not come up (/tmp/esellar-tunnel.log)"
|
|
313
479
|
vm 'podman pull --tls-verify=false 127.0.0.1:5000/esellar-api:latest' || die "app image pull failed"
|
|
314
480
|
else
|
|
315
481
|
say "skipping app-image pre-pull (pass --pull-images, or let the first deploy pull)"
|
|
@@ -349,6 +515,13 @@ vm 'rc-update show default | grep -qE "^[[:space:]]*esellar-api[[:space:]]*\\|"'
|
|
|
349
515
|
vm 'rc-update show default | grep -qE "^[[:space:]]*kamal-proxy[[:space:]]*\\|"' || die "kamal-proxy not in the default runlevel"
|
|
350
516
|
vm 'rc-service kamal-proxy status >/dev/null' || die "kamal-proxy service not started"
|
|
351
517
|
|
|
518
|
+
say "gate: esellar-anchor watcher running + INERT (no anchor conf on a fresh VM)"
|
|
519
|
+
vm 'rc-update show default | grep -qE "^[[:space:]]*esellar-anchor[[:space:]]*\\|"' || die "esellar-anchor not in the default runlevel"
|
|
520
|
+
vm 'rc-service esellar-anchor status >/dev/null' || die "esellar-anchor service not started"
|
|
521
|
+
vm '! test -e /etc/esellar/anchor.conf' || die "/etc/esellar/anchor.conf already exists on a fresh VM (wrong machine?)"
|
|
522
|
+
vm 'grep -q "no anchor conf" /var/log/esellar-anchor.log' \
|
|
523
|
+
|| die "esellar-anchor is started but never logged its idle transition (watcher loop not running?)"
|
|
524
|
+
|
|
352
525
|
# --- summary ---------------------------------------------------------------------
|
|
353
526
|
IP="${HOST#*@}"
|
|
354
527
|
SUGGESTED_PROXY_HOST="${IP//./-}.sslip.io"
|