@khorsheed/dsh-ankh-guard 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -0
- package/README.en.md +75 -29
- package/README.i18n.yaml +2 -2
- package/README.md +74 -29
- package/lib/cli.js +2383 -209
- package/lib/client.js +257 -0
- package/lib/exit-agent.js +5 -2
- package/lib/index.js +722 -38
- package/lib/invariant.js +1 -1
- package/lib/preflight-runner.js +125 -47
- package/lib/processes-BjZgJjQr.js +344 -0
- package/lib/restart-context-D6nISh28.js +1245 -0
- package/lib/restart-context-DUyExi9O.js +1245 -0
- package/lib/{state-Dhx9VG44.js → state-4f7yny39.js} +60 -13
- package/lib/state-CZMypGkB.js +323 -0
- package/lib/test-seam-DnvLWTeO.js +119 -0
- package/lib/test-seam-cli.js +24 -0
- package/lib/test-seam-dwvaKjRp.js +459 -0
- package/lib/test-seam.js +2 -0
- package/lib/types/browser-handoff.d.ts +55 -0
- package/lib/types/browser-handoff.js +489 -0
- package/lib/types/cli.d.ts +34 -4
- package/lib/types/cli.js +1487 -225
- package/lib/types/client/index.d.ts +15 -0
- package/lib/types/client/index.js +264 -0
- package/lib/types/deployment-proof.d.ts +24 -0
- package/lib/types/deployment-proof.js +314 -0
- package/lib/types/exit-agent.js +2 -0
- package/lib/types/git.d.ts +12 -3
- package/lib/types/git.js +69 -7
- package/lib/types/index.d.ts +66 -3
- package/lib/types/index.js +157 -39
- package/lib/types/launch-spec.d.ts +263 -0
- package/lib/types/launch-spec.js +823 -0
- package/lib/types/preflight-runner.d.ts +23 -12
- package/lib/types/preflight-runner.js +152 -57
- package/lib/types/processes.d.ts +38 -6
- package/lib/types/processes.js +236 -10
- package/lib/types/restart-context.d.ts +50 -0
- package/lib/types/restart-context.js +106 -0
- package/lib/types/restart-request.d.ts +32 -0
- package/lib/types/restart-request.js +128 -0
- package/lib/types/state-files.d.ts +30 -0
- package/lib/types/state-files.js +55 -0
- package/lib/types/state.d.ts +29 -2
- package/lib/types/state.js +52 -7
- package/lib/types/temp-artifact.d.ts +15 -0
- package/lib/types/temp-artifact.js +17 -0
- package/lib/types/test-seam-cli.d.ts +3 -0
- package/lib/types/test-seam-cli.js +27 -0
- package/lib/types/test-seam.d.ts +55 -0
- package/lib/types/test-seam.js +112 -0
- package/lib/types/transition.d.ts +118 -0
- package/lib/types/transition.js +717 -0
- package/package.json +29 -9
- package/scripts/dsh-watchdog.sh +1388 -80
- package/scripts/install-launchd.sh +43 -5
- package/scripts/install-systemd.sh +43 -5
- package/scripts/on-install.js +1 -1
- package/skills/dsh-self-restart-guard/SKILL.md +38 -12
- package/lib/processes-hCAmwma-.js +0 -127
- package/lib/restart-context-DmnQXNf-.js +0 -421
package/scripts/dsh-watchdog.sh
CHANGED
|
@@ -28,8 +28,12 @@
|
|
|
28
28
|
# WD_HOME=DIR dsh root (default: $DSH_HOME)
|
|
29
29
|
# WD_STATE_DIR=DIR state dir: markers, pidfile, logs (default: <WD_HOME>/state)
|
|
30
30
|
# WD_PORT=N port to own (default 3080)
|
|
31
|
-
# WD_REPO=DIR
|
|
31
|
+
# WD_REPO=DIR credential/rollback repository (not the host checkout)
|
|
32
|
+
# WD_HARNESS_ROOT=DIR host checkout exported to the child as DSH_HARNESS
|
|
32
33
|
# WD_START="CMD" shell command that starts the supervised instance
|
|
34
|
+
# WD_PROFILE=NAME the profile the instance boots (default web) — its
|
|
35
|
+
# composition inputs are snapshotted at healthy boots and
|
|
36
|
+
# restored when boot failures originate outside the repo
|
|
33
37
|
# WD_GUARD="CMD" how to invoke the guard CLI (default: dsh-ankh-guard)
|
|
34
38
|
# WD_WAIT_OWNER=1 don't adopt the port; wait for the current owner to exit
|
|
35
39
|
# WD_DELAY=N sleep N seconds before adopting/observing the port
|
|
@@ -38,7 +42,21 @@
|
|
|
38
42
|
# WD_ADOPTION=1 the CLI saw a live owner at supervise time — the first
|
|
39
43
|
# boot is a takeover (report it), not a first-ever boot
|
|
40
44
|
# WD_SUPERVISE=1 write/check the pidfile (one watchdog only)
|
|
41
|
-
# WD_BOOT_TIMEOUT=N seconds to
|
|
45
|
+
# WD_BOOT_TIMEOUT=N seconds to prove application readiness (default 60)
|
|
46
|
+
# WD_TAKEOVER_FROM=P replace this live watchdog's pidfile claim before the
|
|
47
|
+
# old instance is interrupted (launch cutover only)
|
|
48
|
+
# WD_CUTOVER_ID=ID durable launch-cutover receipt transaction
|
|
49
|
+
# WD_CUTOVER_POLICY= restore-previous or wait-for-user (approved pre-stop)
|
|
50
|
+
# WD_CUTOVER_DELAY_SECONDS=N grace after supervisor claim before old-child stop
|
|
51
|
+
# WD_PREVIOUS_CHILD_*=PID/start token authoritative old supervisor child root
|
|
52
|
+
# WD_PREVIOUS_LISTENER_*=PID/start token listener inside that old child tree
|
|
53
|
+
# WD_READY_STABILITY_SECONDS=N unchanged child/listener proof window (default 3)
|
|
54
|
+
# WD_BROWSER_HANDOFF=required|off after an authenticated launch-URL exchange
|
|
55
|
+
# WD_TRANSITION_PLAN_SHA256=SHA-256 signals a prepared filesystem transition
|
|
56
|
+
# bound to the active cutover; the guard CLI reads the
|
|
57
|
+
# durable plan and journal rather than trusting this value
|
|
58
|
+
# WD_BROWSER_HANDOFF_TIMEOUT_SECONDS=N wait for original-tab acknowledgement,
|
|
59
|
+
# then (after fallback open) for fallback acknowledgement
|
|
42
60
|
# WD_TEST_FAKE=1 launch a throwaway http server instead of the instance
|
|
43
61
|
# WD_TEST_BREAK=1 launch a command that always fails (give-up testing)
|
|
44
62
|
#
|
|
@@ -55,6 +73,26 @@ PORT="${WD_PORT:-3080}"
|
|
|
55
73
|
DELAY="${WD_DELAY:-0}"
|
|
56
74
|
BOOT_TIMEOUT="${WD_BOOT_TIMEOUT:-60}"
|
|
57
75
|
REPO="${WD_REPO:-}"
|
|
76
|
+
HARNESS_ROOT="${WD_HARNESS_ROOT:-${DSH_HARNESS:-}}"
|
|
77
|
+
PROFILE="${WD_PROFILE:-web}"
|
|
78
|
+
START_CMD="${WD_START:-}"
|
|
79
|
+
CUTOVER_ID="${WD_CUTOVER_ID:-}"
|
|
80
|
+
CUTOVER_POLICY="${WD_CUTOVER_POLICY:-}"
|
|
81
|
+
CUTOVER_ROLE="${WD_CUTOVER_ROLE:-target}"
|
|
82
|
+
PREVIOUS_START="${WD_PREVIOUS_START:-}"
|
|
83
|
+
PREVIOUS_HOME="${WD_PREVIOUS_HOME:-}"
|
|
84
|
+
PREVIOUS_REPO="${WD_PREVIOUS_REPO:-}"
|
|
85
|
+
PREVIOUS_HARNESS_ROOT="${WD_PREVIOUS_HARNESS_ROOT:-}"
|
|
86
|
+
PREVIOUS_PROFILE="${WD_PREVIOUS_PROFILE:-}"
|
|
87
|
+
PREVIOUS_CHILD_PID="${WD_PREVIOUS_CHILD_PID:-}"
|
|
88
|
+
PREVIOUS_CHILD_START="${WD_PREVIOUS_CHILD_START:-}"
|
|
89
|
+
PREVIOUS_LISTENER_PID="${WD_PREVIOUS_LISTENER_PID:-}"
|
|
90
|
+
PREVIOUS_LISTENER_START="${WD_PREVIOUS_LISTENER_START:-}"
|
|
91
|
+
BROWSER_HANDOFF="${WD_BROWSER_HANDOFF:-off}"
|
|
92
|
+
TARGET_FAILURE_LIMIT="${WD_TARGET_FAILURE_LIMIT:-2}"
|
|
93
|
+
READY_STABILITY_SECONDS="${WD_READY_STABILITY_SECONDS:-3}"
|
|
94
|
+
BROWSER_HANDOFF_TIMEOUT_SECONDS="${WD_BROWSER_HANDOFF_TIMEOUT_SECONDS:-8}"
|
|
95
|
+
TRANSITION_PLAN_SHA256="${WD_TRANSITION_PLAN_SHA256:-}"
|
|
58
96
|
# Every marker, the pidfile, and the attempt log live in ONE state directory:
|
|
59
97
|
# WD_STATE_DIR when the guard CLI names it (its --state-dir), else the
|
|
60
98
|
# conventional <home>/state. Deriving it here as <home>/state while the guard
|
|
@@ -67,44 +105,746 @@ RESTART_MARKER="$STATE_DIR/restart-requested.json"
|
|
|
67
105
|
STOP_MARKER="$STATE_DIR/watchdog-stop"
|
|
68
106
|
PIDFILE="$STATE_DIR/watchdog.pid"
|
|
69
107
|
ATTEMPT_LOG="$STATE_DIR/boot-attempt.log"
|
|
108
|
+
CONTROL_FILE="$STATE_DIR/launch-cutover-control.json"
|
|
109
|
+
CONTROL_ABORT_FILE="$STATE_DIR/launch-cutover-abort.json"
|
|
110
|
+
CONTROL_RESTORE_FILE="$STATE_DIR/launch-cutover-restore-previous.json"
|
|
111
|
+
BROWSER_HANDOFF_REQUEST_FILE="$STATE_DIR/browser-handoff-request.json"
|
|
112
|
+
BROWSER_HANDOFF_ACK_FILE="$STATE_DIR/browser-handoff-ack.json"
|
|
70
113
|
|
|
71
|
-
|
|
114
|
+
# Timestamp every lifecycle line so a durable receipt can be correlated with
|
|
115
|
+
# supervisor/child PIDs across launchd, systemd, and detached CLI restarts.
|
|
116
|
+
wd_log() {
|
|
117
|
+
printf '%s [watchdog] %s\n' "$(date '+%Y-%m-%dT%H:%M:%S%z')" "$*"
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
resolve_tool() {
|
|
121
|
+
local name=$1 candidate
|
|
122
|
+
shift
|
|
123
|
+
for candidate in "$@"; do [ -x "$candidate" ] && { printf '%s' "$candidate"; return 0; }; done
|
|
124
|
+
command -v "$name" 2>/dev/null || return 1
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
# macOS tool sessions commonly omit /usr/sbin from PATH. Listener identity is
|
|
128
|
+
# a safety proof, not optional telemetry, so resolve canonical absolute paths
|
|
129
|
+
# and fail loud when the proof machinery truly is unavailable.
|
|
130
|
+
LSOF_BIN=$(resolve_tool lsof /usr/sbin/lsof /usr/bin/lsof) || { wd_log "lsof is required to prove listener ownership" >&2; exit 1; }
|
|
131
|
+
PS_BIN=$(resolve_tool ps /bin/ps /usr/bin/ps) || { wd_log "ps is required to prove process identity" >&2; exit 1; }
|
|
132
|
+
PGREP_BIN=$(resolve_tool pgrep /usr/bin/pgrep /bin/pgrep) || { wd_log "pgrep is required to manage the supervised child tree" >&2; exit 1; }
|
|
133
|
+
SYSCTL_BIN=$(resolve_tool sysctl /usr/sbin/sysctl /sbin/sysctl 2>/dev/null || true)
|
|
134
|
+
PYTHON_BIN=$(resolve_tool python3 /usr/bin/python3 /opt/homebrew/bin/python3 2>/dev/null || true)
|
|
135
|
+
|
|
136
|
+
[ -n "$DSH_ROOT" ] || { wd_log "WD_HOME or DSH_HOME must be set" >&2; exit 1; }
|
|
72
137
|
export DSH_HOME="$DSH_ROOT"
|
|
73
138
|
mkdir -p "$STATE_DIR"
|
|
74
|
-
|
|
139
|
+
|
|
140
|
+
# Production sleeps retain their exact durations. A test run may scale only
|
|
141
|
+
# internal polling/backoff after presenting the private run coordinates used
|
|
142
|
+
# by the ownership ledger; the scale is never persisted in a launch spec.
|
|
143
|
+
wd_sleep() {
|
|
144
|
+
local duration=$1 scale=${ANKH_GUARD_TEST_SLEEP_SCALE:-1}
|
|
145
|
+
if [ -n "${ANKH_GUARD_TEST_RUN_DIR:-}" ] && [ -n "${ANKH_GUARD_TEST_RUN_TOKEN:-}" ] \
|
|
146
|
+
&& printf '%s' "$scale" | grep -Eq '^0\.[0-9]+$|^1(\.0+)?$'; then
|
|
147
|
+
duration=$(/usr/bin/awk -v duration="$duration" -v scale="$scale" \
|
|
148
|
+
'BEGIN { value=duration*scale; if (value < 0.01) value=0.01; printf "%.3f", value }')
|
|
149
|
+
fi
|
|
150
|
+
sleep "$duration"
|
|
151
|
+
}
|
|
75
152
|
|
|
76
153
|
launch_instance() {
|
|
77
|
-
|
|
154
|
+
test_register_self instance-wrapper
|
|
155
|
+
test_event_self instance-wrapper process-started
|
|
156
|
+
if [ "${WD_TEST_BREAK:-0}" = "1" ]; then wd_sleep 1; exit 1; fi
|
|
78
157
|
if [ "${WD_TEST_FAKE:-0}" = "1" ]; then
|
|
79
158
|
node -e "require('http').createServer((q,s)=>s.end('ok')).listen($PORT,'127.0.0.1')"
|
|
80
159
|
exit
|
|
81
160
|
fi
|
|
82
|
-
if [ -z "$
|
|
83
|
-
|
|
161
|
+
if [ -z "$START_CMD" ]; then wd_log "launch command unset — nothing to supervise" >&2; exit 1; fi
|
|
162
|
+
# The instance inherits this process's environment: scrub EVERY WD_* so no
|
|
163
|
+
# supervision variable can leak into the shells the instance hosts. A leaked
|
|
164
|
+
# WD_STATE_DIR retargets any watchdog script those shells spawn (observed
|
|
165
|
+
# 2026-08-29: an agent session inside the supervised deployment ran the test
|
|
166
|
+
# suite straight into the PROD state dir — racers yielded to the live
|
|
167
|
+
# pidfile owner, and the reclaim case never saw its temp pidfile). The scrub
|
|
168
|
+
# is prefix-based, not a name list: the CLI adds WD_* variables over time
|
|
169
|
+
# (WD_GUARD, WD_WAIT_OWNER, WD_ADOPTION, …) and a list silently goes stale.
|
|
170
|
+
# WD_START is captured first: the unset would otherwise eat the command
|
|
171
|
+
# itself. The guard CLI's bare-restart spawn applies the same scrub.
|
|
172
|
+
local start_cmd=$START_CMD launch_home=$DSH_ROOT launch_harness_root=$HARNESS_ROOT
|
|
173
|
+
(
|
|
174
|
+
export DSH_HOME="$launch_home"
|
|
175
|
+
if [ -n "$launch_harness_root" ]; then export DSH_HARNESS="$launch_harness_root"; fi
|
|
176
|
+
cd "$launch_home/home" 2>/dev/null || cd /tmp || exit 1
|
|
177
|
+
for v in $(env | sed -n 's/^\(WD_[^=]*\)=.*/\1/p'); do unset "$v"; done
|
|
178
|
+
sh -c "$start_cmd"
|
|
179
|
+
)
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
http_status() {
|
|
183
|
+
curl -s --noproxy '*' -o /dev/null -w '%{http_code}' --max-time 3 "http://127.0.0.1:$PORT/" 2>/dev/null || true
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
cutover_event() {
|
|
187
|
+
[ -n "$CUTOVER_ID" ] || return 0
|
|
188
|
+
guard_cmd cutover-event "$CUTOVER_ID" "$@" --state-dir "$STATE_DIR" >/dev/null 2>&1
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
# Receipt updates are part of the transaction, not telemetry. Keep the proven
|
|
192
|
+
# child (or the still-running old child during supervisor handoff) available
|
|
193
|
+
# while retrying a transient state/CLI failure; never advance in memory past a
|
|
194
|
+
# durable event that crash recovery depends on.
|
|
195
|
+
cutover_event_required() {
|
|
196
|
+
[ -n "$CUTOVER_ID" ] || return 0
|
|
197
|
+
while ! cutover_event "$@"; do
|
|
198
|
+
wd_log "could not persist cutover event $1 — retrying; service state is unchanged" >&2
|
|
199
|
+
wd_sleep 1
|
|
200
|
+
done
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
cutover_control_action() {
|
|
204
|
+
[ -n "$CUTOVER_ID" ] || return 0
|
|
205
|
+
node -e '
|
|
206
|
+
const fs = require("fs")
|
|
207
|
+
const [restoreFile, abortFile, legacyFile, id] = process.argv.slice(1)
|
|
208
|
+
const read = (file) => {
|
|
209
|
+
try {
|
|
210
|
+
const value = JSON.parse(fs.readFileSync(file, "utf8"))
|
|
211
|
+
return value?.version === 1 && value.cutoverId === id
|
|
212
|
+
&& (value.action === "abort" || value.action === "restore-previous") ? value.action : ""
|
|
213
|
+
} catch { return "" }
|
|
214
|
+
}
|
|
215
|
+
const restore = read(restoreFile)
|
|
216
|
+
const legacy = read(legacyFile)
|
|
217
|
+
process.stdout.write(restore === "restore-previous" || legacy === "restore-previous"
|
|
218
|
+
? "restore-previous" : read(abortFile) || legacy)
|
|
219
|
+
' "$CONTROL_RESTORE_FILE" "$CONTROL_ABORT_FILE" "$CONTROL_FILE" "$CUTOVER_ID" 2>/dev/null
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
# The host owns the shape of its per-process launch URL. Discovery is generic:
|
|
223
|
+
# the first HTTP URL printed by THIS attempt whose authority is exactly the
|
|
224
|
+
# supervised loopback authority and whose query is non-empty. No parameter
|
|
225
|
+
# name ("token" or otherwise) is part of the watchdog protocol.
|
|
226
|
+
launch_url_from_output() {
|
|
227
|
+
node -e '
|
|
228
|
+
const fs = require("fs")
|
|
229
|
+
const [file, port] = process.argv.slice(1)
|
|
230
|
+
let text = ""
|
|
231
|
+
try { text = fs.readFileSync(file, "utf8") } catch {}
|
|
232
|
+
for (const match of text.matchAll(/https?:\/\/[^\s)]+/g)) {
|
|
233
|
+
try {
|
|
234
|
+
const url = new URL(match[0])
|
|
235
|
+
if (url.protocol === "http:" && url.hostname === "127.0.0.1"
|
|
236
|
+
&& url.port === port && url.pathname === "/" && url.search !== ""
|
|
237
|
+
&& url.username === "" && url.password === "" && url.hash === "") {
|
|
238
|
+
process.stdout.write(url.href)
|
|
239
|
+
break
|
|
240
|
+
}
|
|
241
|
+
} catch {}
|
|
242
|
+
}
|
|
243
|
+
' "$ATTEMPT_LOG" "$PORT" 2>/dev/null
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
# The process output is durable operational evidence, but a launch URL is a
|
|
247
|
+
# bearer credential. Once captured in memory, overwrite every matching URL in
|
|
248
|
+
# place with an equal-length marker. Equal length preserves the active child's
|
|
249
|
+
# append offset; an atomic rename here would strand later output on an unlinked
|
|
250
|
+
# inode. The failure path calls this too, so a process that prints then exits
|
|
251
|
+
# cannot have its credential mirrored into the watchdog log.
|
|
252
|
+
redact_launch_urls_in_output() {
|
|
253
|
+
node -e '
|
|
254
|
+
const fs = require("fs")
|
|
255
|
+
const [file, port] = process.argv.slice(1)
|
|
256
|
+
let text
|
|
257
|
+
try { text = fs.readFileSync(file, "utf8") } catch { process.exit(0) }
|
|
258
|
+
const edits = []
|
|
259
|
+
for (const match of text.matchAll(/https?:\/\/[^\s)]+/g)) {
|
|
260
|
+
try {
|
|
261
|
+
const url = new URL(match[0])
|
|
262
|
+
if (url.protocol === "http:" && url.hostname === "127.0.0.1"
|
|
263
|
+
&& url.port === port && url.pathname === "/" && url.search !== ""
|
|
264
|
+
&& url.username === "" && url.password === "" && url.hash === "") {
|
|
265
|
+
const start = Buffer.byteLength(text.slice(0, match.index))
|
|
266
|
+
const length = Buffer.byteLength(match[0])
|
|
267
|
+
edits.push({ start, length })
|
|
268
|
+
}
|
|
269
|
+
} catch {}
|
|
270
|
+
}
|
|
271
|
+
if (edits.length === 0) process.exit(0)
|
|
272
|
+
const fd = fs.openSync(file, "r+")
|
|
273
|
+
try {
|
|
274
|
+
for (const { start, length } of edits) {
|
|
275
|
+
const label = Buffer.from("[launch-url-redacted]")
|
|
276
|
+
const replacement = Buffer.alloc(length, 0x20)
|
|
277
|
+
label.copy(replacement, 0, 0, Math.min(label.length, replacement.length))
|
|
278
|
+
fs.writeSync(fd, replacement, 0, replacement.length, start)
|
|
279
|
+
}
|
|
280
|
+
} finally { fs.closeSync(fd) }
|
|
281
|
+
' "$ATTEMPT_LOG" "$PORT" 2>/dev/null || true
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
open_launch_url() {
|
|
285
|
+
local url=$1
|
|
286
|
+
if [ -n "${WD_BROWSER_OPEN_COMMAND:-}" ]; then
|
|
287
|
+
"${WD_BROWSER_OPEN_COMMAND}" "$url" >/dev/null 2>&1
|
|
288
|
+
elif command -v open >/dev/null 2>&1; then
|
|
289
|
+
open "$url" >/dev/null 2>&1
|
|
290
|
+
elif command -v xdg-open >/dev/null 2>&1; then
|
|
291
|
+
xdg-open "$url" >/dev/null 2>&1
|
|
292
|
+
elif command -v gio >/dev/null 2>&1; then
|
|
293
|
+
gio open "$url" >/dev/null 2>&1
|
|
294
|
+
else
|
|
295
|
+
return 1
|
|
296
|
+
fi
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
last_transport_status=''
|
|
300
|
+
launch_url_reported=0
|
|
301
|
+
launch_url_value=''
|
|
302
|
+
browser_handoff_done=0
|
|
303
|
+
browser_handoff_reported=0
|
|
304
|
+
handoff_cookie_jar=''
|
|
305
|
+
readiness_detail=''
|
|
306
|
+
protected_ready=0
|
|
307
|
+
|
|
308
|
+
# Readiness has two layers. Any HTTP response proves transport-up; only a bare
|
|
309
|
+
# 200, or a same-authority launch-URL exchange (303 + cookie-authenticated 200)
|
|
310
|
+
# proves application readiness. A naked 401 therefore never counts as ready.
|
|
311
|
+
ready_probe() {
|
|
312
|
+
local status url jar exchange authenticated
|
|
313
|
+
current_owned_listener || return 1
|
|
314
|
+
# The test fixture records the proven runtime identity as soon as ownership
|
|
315
|
+
# is established. This is not production discovery or port-based cleanup:
|
|
316
|
+
# teardown later signals only this immutable PID/start-token lease.
|
|
317
|
+
test_register_pid "$current_listener_pid" instance-listener
|
|
318
|
+
status=$(http_status)
|
|
319
|
+
current_ownership_matches || return 1
|
|
320
|
+
if [ -n "$status" ] && [ "$status" != "000" ] && [ "$status" != "$last_transport_status" ]; then
|
|
321
|
+
last_transport_status=$status
|
|
322
|
+
cutover_event_required transport "$status"
|
|
323
|
+
if [ "$status" != "200" ]; then
|
|
324
|
+
wd_log "transport up on :$PORT (HTTP $status); application readiness still pending"
|
|
325
|
+
fi
|
|
326
|
+
fi
|
|
327
|
+
if [ "$status" = "200" ]; then
|
|
328
|
+
protected_ready=0
|
|
329
|
+
readiness_detail="plain HTTP 200"
|
|
330
|
+
return 0
|
|
331
|
+
fi
|
|
332
|
+
|
|
333
|
+
if [ -z "$launch_url_value" ]; then
|
|
334
|
+
launch_url_value=$(launch_url_from_output)
|
|
335
|
+
[ -n "$launch_url_value" ] && redact_launch_urls_in_output
|
|
336
|
+
fi
|
|
337
|
+
url=$launch_url_value
|
|
338
|
+
[ -n "$url" ] || return 1
|
|
339
|
+
if [ "$launch_url_reported" = "0" ]; then
|
|
340
|
+
wd_log "observed same-authority launch URL (credential redacted)"
|
|
341
|
+
cutover_event_required launch-url
|
|
342
|
+
launch_url_reported=1
|
|
343
|
+
fi
|
|
344
|
+
jar=$(mktemp "${TMPDIR:-/tmp}/ankh-guard-handoff.XXXXXX") || return 1
|
|
345
|
+
handoff_cookie_jar=$jar
|
|
346
|
+
chmod 600 "$jar" 2>/dev/null || true
|
|
347
|
+
exchange=$(curl -sS --noproxy '*' -c "$jar" -o /dev/null -w '%{http_code}' --max-time 3 "$url" 2>/dev/null || true)
|
|
348
|
+
cutover_event_required auth-exchange "${exchange:-0}"
|
|
349
|
+
if [ "$exchange" != "303" ]; then rm -f "$jar"; handoff_cookie_jar=''; return 1; fi
|
|
350
|
+
authenticated=$(curl -sS --noproxy '*' -b "$jar" -o /dev/null -w '%{http_code}' --max-time 3 "http://127.0.0.1:$PORT/" 2>/dev/null || true)
|
|
351
|
+
rm -f "$jar"
|
|
352
|
+
handoff_cookie_jar=''
|
|
353
|
+
cutover_event_required authenticated "${authenticated:-0}"
|
|
354
|
+
[ "$authenticated" = "200" ] || return 1
|
|
355
|
+
current_ownership_matches || return 1
|
|
356
|
+
protected_ready=1
|
|
357
|
+
readiness_detail="authenticated launch URL: 303 exchange, cookie / = 200"
|
|
358
|
+
return 0
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
# Whether the previous, proven listener armed an original tab for this
|
|
362
|
+
# transaction. All registered tabs are eligible to recover; the file contains
|
|
363
|
+
# only capability digests, never raw capabilities or bearer launch URLs.
|
|
364
|
+
browser_original_registered() {
|
|
365
|
+
node -e '
|
|
366
|
+
const fs = require("fs")
|
|
367
|
+
try {
|
|
368
|
+
const value = JSON.parse(fs.readFileSync(process.argv[1], "utf8"))
|
|
369
|
+
if (value.version !== 2 || value.cutoverId !== process.argv[2]
|
|
370
|
+
|| !Array.isArray(value.registrations) || value.registrations.length < 1
|
|
371
|
+
|| value.registrations.length > 64) process.exit(1)
|
|
372
|
+
let expectedPort = false
|
|
373
|
+
for (const registration of value.registrations) {
|
|
374
|
+
const authority = new URL(`http://${registration.authority ?? ""}`)
|
|
375
|
+
if (typeof registration.authority !== "string" || registration.authority === ""
|
|
376
|
+
|| authority.host !== registration.authority
|
|
377
|
+
|| authority.username !== "" || authority.password !== ""
|
|
378
|
+
|| !/^[a-f0-9]{64}$/.test(registration.capabilitySha256)
|
|
379
|
+
|| !Number.isFinite(registration.armedAt)) process.exit(1)
|
|
380
|
+
if (authority.port === process.argv[3]) expectedPort = true
|
|
381
|
+
}
|
|
382
|
+
if (!expectedPort) process.exit(1)
|
|
383
|
+
} catch { process.exit(1) }
|
|
384
|
+
' "$BROWSER_HANDOFF_REQUEST_FILE" "$CUTOVER_ID" "$PORT" >/dev/null 2>&1
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
# Print non-secret acknowledgement evidence only when it names this exact
|
|
388
|
+
# stable listener identity and cutover role.
|
|
389
|
+
browser_ack_evidence() {
|
|
390
|
+
node -e '
|
|
391
|
+
const fs = require("fs")
|
|
392
|
+
try {
|
|
393
|
+
const value = JSON.parse(fs.readFileSync(process.argv[1], "utf8"))
|
|
394
|
+
const authority = new URL(`http://${value.authority ?? ""}`)
|
|
395
|
+
if (value.version !== 1 || value.cutoverId !== process.argv[2]
|
|
396
|
+
|| value.role !== process.argv[3] || String(value.listenerPid) !== process.argv[4]
|
|
397
|
+
|| value.listenerStartToken !== process.argv[5]
|
|
398
|
+
|| !["original-tab", "fallback-tab"].includes(value.channel)
|
|
399
|
+
|| !["existing-cookie", "launch-url"].includes(value.authentication)
|
|
400
|
+
|| typeof value.authority !== "string" || value.authority === ""
|
|
401
|
+
|| authority.host !== value.authority || authority.port !== process.argv[6]
|
|
402
|
+
|| authority.username !== "" || authority.password !== ""
|
|
403
|
+
|| !Number.isFinite(value.acknowledgedAt) || value.acknowledgedAt <= 0
|
|
404
|
+
|| value.authority.includes("|")) process.exit(1)
|
|
405
|
+
process.stdout.write(`${value.channel}|${value.authentication}|${value.authority}`)
|
|
406
|
+
} catch { process.exit(1) }
|
|
407
|
+
' "$BROWSER_HANDOFF_ACK_FILE" "$CUTOVER_ID" "$CUTOVER_ROLE" \
|
|
408
|
+
"$current_listener_pid" "$current_listener_start" "$PORT" 2>/dev/null
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
wait_for_browser_ack() {
|
|
412
|
+
local deadline evidence rest
|
|
413
|
+
case "$BROWSER_HANDOFF_TIMEOUT_SECONDS" in ''|*[!0-9]*) return 1 ;; esac
|
|
414
|
+
[ "$BROWSER_HANDOFF_TIMEOUT_SECONDS" -gt 0 ] || return 1
|
|
415
|
+
deadline=$(( $(now_ms) + BROWSER_HANDOFF_TIMEOUT_SECONDS * 1000 ))
|
|
416
|
+
while [ "$(now_ms)" -lt "$deadline" ]; do
|
|
417
|
+
[ -z "$(cutover_control_action)" ] || return 1
|
|
418
|
+
current_ownership_matches || return 1
|
|
419
|
+
evidence=$(browser_ack_evidence) || evidence=''
|
|
420
|
+
if [ -n "$evidence" ]; then
|
|
421
|
+
browser_ack_channel=${evidence%%|*}
|
|
422
|
+
rest=${evidence#*|}
|
|
423
|
+
browser_ack_authentication=${rest%%|*}
|
|
424
|
+
browser_ack_authority=${rest#*|}
|
|
425
|
+
return 0
|
|
426
|
+
fi
|
|
427
|
+
wd_sleep 0.25
|
|
428
|
+
done
|
|
429
|
+
return 1
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
# Browser handoff is deliberately after the ownership stability window and
|
|
433
|
+
# canary. An opener's exit status proves only that a fallback was attempted;
|
|
434
|
+
# readiness requires a page-authored acknowledgement naming the stable listener.
|
|
435
|
+
complete_browser_handoff() {
|
|
436
|
+
local fallback_url
|
|
437
|
+
current_ownership_matches || return 1
|
|
438
|
+
[ "$protected_ready" = "1" ] || return 0
|
|
439
|
+
if [ "$BROWSER_HANDOFF" = "required" ] && [ "$browser_handoff_done" = "0" ]; then
|
|
440
|
+
[ -n "$launch_url_value" ] || return 1
|
|
441
|
+
if browser_original_registered; then
|
|
442
|
+
wd_log "waiting for an armed original browser tab to acknowledge the final process"
|
|
443
|
+
if wait_for_browser_ack; then
|
|
444
|
+
browser_handoff_done=1
|
|
445
|
+
browser_handoff_reported=1
|
|
446
|
+
cutover_event_required browser-handoff acknowledged "$browser_ack_channel" \
|
|
447
|
+
"$browser_ack_authentication" "$browser_ack_authority"
|
|
448
|
+
wd_log "original browser tab acknowledged handoff ($browser_ack_authentication; authority $browser_ack_authority); all registered responsive tabs remain eligible to recover"
|
|
449
|
+
else
|
|
450
|
+
wd_log "original browser tab did not acknowledge within ${BROWSER_HANDOFF_TIMEOUT_SECONDS}s; falling back to system open"
|
|
451
|
+
fi
|
|
452
|
+
else
|
|
453
|
+
wd_log "no original browser tab registered before shutdown; falling back to system open"
|
|
454
|
+
fi
|
|
455
|
+
if [ "$browser_handoff_done" = "0" ]; then
|
|
456
|
+
fallback_url="${launch_url_value}#ankh-guard-handoff=${CUTOVER_ID}"
|
|
457
|
+
if open_launch_url "$fallback_url"; then
|
|
458
|
+
cutover_event_required browser-fallback-opened
|
|
459
|
+
wd_log "browser fallback open requested; waiting for page acknowledgement"
|
|
460
|
+
if wait_for_browser_ack; then
|
|
461
|
+
browser_handoff_done=1
|
|
462
|
+
browser_handoff_reported=1
|
|
463
|
+
cutover_event_required browser-handoff acknowledged "$browser_ack_channel" \
|
|
464
|
+
"$browser_ack_authentication" "$browser_ack_authority"
|
|
465
|
+
wd_log "browser acknowledged fallback handoff ($browser_ack_channel; authority $browser_ack_authority)"
|
|
466
|
+
fi
|
|
467
|
+
fi
|
|
468
|
+
fi
|
|
469
|
+
if [ "$browser_handoff_done" = "0" ]; then
|
|
470
|
+
if [ "$browser_handoff_reported" = "0" ]; then
|
|
471
|
+
wd_log "browser handoff failed: no page acknowledgement — not ready"
|
|
472
|
+
cutover_event_required browser-handoff failed
|
|
473
|
+
browser_handoff_reported=1
|
|
474
|
+
fi
|
|
475
|
+
return 1
|
|
476
|
+
fi
|
|
477
|
+
elif [ "$BROWSER_HANDOFF" = "off" ] && [ "$browser_handoff_reported" = "0" ]; then
|
|
478
|
+
cutover_event_required browser-handoff off
|
|
479
|
+
browser_handoff_reported=1
|
|
480
|
+
fi
|
|
481
|
+
current_ownership_matches || return 1
|
|
482
|
+
if [ "$browser_handoff_done" = "1" ]; then
|
|
483
|
+
readiness_detail="authenticated launch URL: 303 exchange, cookie / = 200, browser page acknowledged"
|
|
484
|
+
fi
|
|
485
|
+
return 0
|
|
486
|
+
}
|
|
487
|
+
|
|
488
|
+
# Initial HTTP/auth readiness is provisional. Hold the exact child PID/start
|
|
489
|
+
# identity and exact listener PID/start identity unchanged for a stability
|
|
490
|
+
# window, with no retry tolerated inside that window, before canary/terminal
|
|
491
|
+
# receipt. A child that exits after first returning 200 therefore fails the
|
|
492
|
+
# cutover instead of borrowing another process's response.
|
|
493
|
+
prove_stable_readiness() {
|
|
494
|
+
local expected_child=$child expected_child_start=$child_start_token
|
|
495
|
+
local expected_listener=$current_listener_pid expected_listener_start=$current_listener_start
|
|
496
|
+
local deadline
|
|
497
|
+
case "$READY_STABILITY_SECONDS" in ''|*[!0-9]*) return 1 ;; esac
|
|
498
|
+
[ "$READY_STABILITY_SECONDS" -gt 0 ] || return 1
|
|
499
|
+
deadline=$(( $(now_ms) + READY_STABILITY_SECONDS * 1000 ))
|
|
500
|
+
while [ "$(now_ms)" -lt "$deadline" ]; do
|
|
501
|
+
[ -z "$(cutover_control_action)" ] || return 1
|
|
502
|
+
[ "$child" = "$expected_child" ] && [ "$child_start_token" = "$expected_child_start" ] || return 1
|
|
503
|
+
current_listener_pid=$expected_listener
|
|
504
|
+
current_listener_start=$expected_listener_start
|
|
505
|
+
current_ownership_matches || return 1
|
|
506
|
+
ready_probe || return 1
|
|
507
|
+
[ "$current_listener_pid" = "$expected_listener" ] \
|
|
508
|
+
&& [ "$current_listener_start" = "$expected_listener_start" ] || return 1
|
|
509
|
+
wd_sleep 0.25
|
|
510
|
+
done
|
|
511
|
+
current_listener_pid=$expected_listener
|
|
512
|
+
current_listener_start=$expected_listener_start
|
|
513
|
+
current_ownership_matches || return 1
|
|
514
|
+
readiness_detail="$readiness_detail; ownership stable ${READY_STABILITY_SECONDS}s (child $child, listener $current_listener_pid, retry 0)"
|
|
515
|
+
cutover_event_required ownership-stable "$CUTOVER_ROLE" "$child" "$child_start_token" \
|
|
516
|
+
"$current_listener_pid" "$current_listener_start" "$((READY_STABILITY_SECONDS * 1000))" 0
|
|
517
|
+
return 0
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
transport_up() {
|
|
521
|
+
local status
|
|
522
|
+
status=$(http_status)
|
|
523
|
+
[ -n "$status" ] && [ "$status" != "000" ]
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
process_start_token() {
|
|
527
|
+
if [ "$(uname -s 2>/dev/null)" = "Darwin" ] && [ -n "$PYTHON_BIN" ]; then
|
|
528
|
+
"$PYTHON_BIN" -c 'import ctypes,struct,sys;p=int(sys.argv[1]);b=ctypes.create_string_buffer(136);n=ctypes.CDLL("/usr/lib/libproc.dylib").proc_pidinfo(p,3,0,b,136);n == 136 or sys.exit(1);s,u=struct.unpack_from("QQ",b.raw,120);print(f"darwin:{s}:{u}",end="")' "$1" 2>/dev/null && return 0
|
|
529
|
+
fi
|
|
530
|
+
PROCESS_PS="$PS_BIN" PROCESS_SYSCTL="$SYSCTL_BIN" node -e '
|
|
531
|
+
const { createHash } = require("crypto")
|
|
532
|
+
const { existsSync, readFileSync } = require("fs")
|
|
533
|
+
const { execFileSync } = require("child_process")
|
|
534
|
+
const pid = Number(process.argv[1])
|
|
535
|
+
try {
|
|
536
|
+
const run = (file, args) => execFileSync(file, args, { encoding: "utf8", stdio: "pipe" }).trim()
|
|
537
|
+
const status = run(process.env.PROCESS_PS, ["-o", "stat=", "-p", String(pid)])
|
|
538
|
+
if (!status || status.startsWith("Z")) process.exit(1)
|
|
539
|
+
const procStat = `/proc/${pid}/stat`
|
|
540
|
+
if (existsSync(procStat)) {
|
|
541
|
+
const stat = readFileSync(procStat, "utf8")
|
|
542
|
+
const end = stat.lastIndexOf(")")
|
|
543
|
+
const fields = end < 0 ? [] : stat.slice(end + 1).trim().split(/\s+/)
|
|
544
|
+
const ticks = fields[19]
|
|
545
|
+
if (!ticks) process.exit(1)
|
|
546
|
+
let boot = "unknown-boot"
|
|
547
|
+
try { boot = readFileSync("/proc/sys/kernel/random/boot_id", "utf8").trim() || boot } catch {}
|
|
548
|
+
process.stdout.write(`linux:${boot}:${ticks}`)
|
|
549
|
+
process.exit(0)
|
|
550
|
+
}
|
|
551
|
+
const base = run(process.env.PROCESS_PS, ["-o", "sess=", "-o", "uid=", "-o", "lstart=", "-o", "command=", "-p", String(pid)])
|
|
552
|
+
if (!base) process.exit(1)
|
|
553
|
+
let boot = "unknown-boot"
|
|
554
|
+
try { if (process.env.PROCESS_SYSCTL) boot = run(process.env.PROCESS_SYSCTL, ["-n", "kern.boottime"]) || boot } catch {}
|
|
555
|
+
process.stdout.write("posix:" + createHash("sha256").update(boot).update("\0").update(base).digest("hex"))
|
|
556
|
+
} catch { process.exit(1) }
|
|
557
|
+
' "$1" 2>/dev/null
|
|
558
|
+
}
|
|
559
|
+
|
|
560
|
+
# Test-only birth registration and event sink. The shipped watchdog is inert
|
|
561
|
+
# unless a fixture provides an explicit private run directory, token, and the
|
|
562
|
+
# built registrar path. Each shell/Node child initiates its own record so the
|
|
563
|
+
# outer Vitest process never has to discover a replacement through a mutable
|
|
564
|
+
# pidfile. Registration failure is diagnostic-only and cannot alter production
|
|
565
|
+
# supervision behavior.
|
|
566
|
+
test_register_pid() {
|
|
567
|
+
[ -n "${ANKH_GUARD_TEST_RUN_DIR:-}" ] || return 0
|
|
568
|
+
[ -n "${ANKH_GUARD_TEST_RUN_TOKEN:-}" ] || return 0
|
|
569
|
+
[ -f "${ANKH_GUARD_TEST_REGISTER_BIN:-}" ] || return 0
|
|
570
|
+
ANKH_GUARD_TEST_PROCESS_ROLE="$2" node "$ANKH_GUARD_TEST_REGISTER_BIN" register-pid "$1" "$2" >/dev/null 2>&1 || true
|
|
571
|
+
}
|
|
572
|
+
|
|
573
|
+
test_register_self() {
|
|
574
|
+
[ -n "${ANKH_GUARD_TEST_RUN_DIR:-}" ] || return 0
|
|
575
|
+
[ -n "${ANKH_GUARD_TEST_RUN_TOKEN:-}" ] || return 0
|
|
576
|
+
[ -f "${ANKH_GUARD_TEST_REGISTER_BIN:-}" ] || return 0
|
|
577
|
+
ANKH_GUARD_TEST_PROCESS_ROLE="$1" node "$ANKH_GUARD_TEST_REGISTER_BIN" register-parent "$1" >/dev/null 2>&1 || true
|
|
578
|
+
}
|
|
579
|
+
|
|
580
|
+
test_event_pid() {
|
|
581
|
+
[ -n "${ANKH_GUARD_TEST_RUN_DIR:-}" ] || return 0
|
|
582
|
+
[ -n "${ANKH_GUARD_TEST_RUN_TOKEN:-}" ] || return 0
|
|
583
|
+
[ -f "${ANKH_GUARD_TEST_REGISTER_BIN:-}" ] || return 0
|
|
584
|
+
ANKH_GUARD_TEST_PROCESS_ROLE="$2" node "$ANKH_GUARD_TEST_REGISTER_BIN" event-pid "$1" "$2" "$3" >/dev/null 2>&1 || true
|
|
585
|
+
}
|
|
586
|
+
|
|
587
|
+
test_event_self() {
|
|
588
|
+
[ -n "${ANKH_GUARD_TEST_RUN_DIR:-}" ] || return 0
|
|
589
|
+
[ -n "${ANKH_GUARD_TEST_RUN_TOKEN:-}" ] || return 0
|
|
590
|
+
[ -f "${ANKH_GUARD_TEST_REGISTER_BIN:-}" ] || return 0
|
|
591
|
+
ANKH_GUARD_TEST_PROCESS_ROLE="$1" node "$ANKH_GUARD_TEST_REGISTER_BIN" event-parent "$1" "$2" >/dev/null 2>&1 || true
|
|
592
|
+
}
|
|
593
|
+
|
|
594
|
+
now_ms() {
|
|
595
|
+
node -e 'process.stdout.write(String(Date.now()))'
|
|
596
|
+
}
|
|
597
|
+
|
|
598
|
+
identity_matches() {
|
|
599
|
+
local pid=$1 expected=$2 actual
|
|
600
|
+
[ -n "$pid" ] && [ -n "$expected" ] || return 1
|
|
601
|
+
actual=$(process_start_token "$pid")
|
|
602
|
+
[ -n "$actual" ] && [ "$actual" = "$expected" ]
|
|
603
|
+
}
|
|
604
|
+
|
|
605
|
+
listener_pids() {
|
|
606
|
+
"$LSOF_BIN" -tiTCP:"$PORT" -sTCP:LISTEN -P 2>/dev/null | sort -u
|
|
84
607
|
}
|
|
85
608
|
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
[ "$
|
|
609
|
+
pid_is_listener() {
|
|
610
|
+
local expected=$1 candidate
|
|
611
|
+
for candidate in $(listener_pids); do [ "$candidate" = "$expected" ] && return 0; done
|
|
612
|
+
return 1
|
|
613
|
+
}
|
|
614
|
+
|
|
615
|
+
pid_belongs_to_tree() {
|
|
616
|
+
local root=$1 cursor=$2 parent hops=0
|
|
617
|
+
while [ "$cursor" -gt 0 ] 2>/dev/null && [ "$hops" -lt 256 ]; do
|
|
618
|
+
[ "$cursor" = "$root" ] && return 0
|
|
619
|
+
parent=$("$PS_BIN" -o ppid= -p "$cursor" 2>/dev/null | tr -d ' ')
|
|
620
|
+
[ -n "$parent" ] || return 1
|
|
621
|
+
cursor=$parent
|
|
622
|
+
hops=$((hops + 1))
|
|
623
|
+
done
|
|
624
|
+
return 1
|
|
625
|
+
}
|
|
626
|
+
|
|
627
|
+
# Prove one unchanged listener inside the launched child tree. The response
|
|
628
|
+
# probe is accepted only while this identity remains true before and after
|
|
629
|
+
# the HTTP exchange, preventing a stale/foreign server on the same port from
|
|
630
|
+
# being mistaken for the target.
|
|
631
|
+
current_owned_listener() {
|
|
632
|
+
local pids count listener token
|
|
633
|
+
identity_matches "$child" "$child_start_token" || return 1
|
|
634
|
+
pids=$(listener_pids)
|
|
635
|
+
count=$(printf '%s\n' "$pids" | sed '/^$/d' | wc -l | tr -d ' ')
|
|
636
|
+
[ "$count" = "1" ] || return 1
|
|
637
|
+
listener=$(printf '%s\n' "$pids" | head -1)
|
|
638
|
+
pid_belongs_to_tree "$child" "$listener" || return 1
|
|
639
|
+
token=$(process_start_token "$listener")
|
|
640
|
+
[ -n "$token" ] || return 1
|
|
641
|
+
current_listener_pid=$listener
|
|
642
|
+
current_listener_start=$token
|
|
643
|
+
return 0
|
|
644
|
+
}
|
|
645
|
+
|
|
646
|
+
current_ownership_matches() {
|
|
647
|
+
local pids count token
|
|
648
|
+
identity_matches "$child" "$child_start_token" || return 1
|
|
649
|
+
pids=$(listener_pids)
|
|
650
|
+
count=$(printf '%s\n' "$pids" | sed '/^$/d' | wc -l | tr -d ' ')
|
|
651
|
+
[ "$count" = "1" ] || return 1
|
|
652
|
+
[ "$(printf '%s\n' "$pids" | head -1)" = "$current_listener_pid" ] || return 1
|
|
653
|
+
token=$(process_start_token "$current_listener_pid")
|
|
654
|
+
[ -n "$token" ] && [ "$token" = "$current_listener_start" ] \
|
|
655
|
+
&& pid_belongs_to_tree "$child" "$current_listener_pid"
|
|
89
656
|
}
|
|
90
657
|
|
|
91
658
|
# Reap a pid AND its descendants, deepest first (best effort). The watchdog
|
|
92
659
|
# guarantees the direct child; the sweep keeps grandchildren from outliving
|
|
93
660
|
# the instance — a single-pid kill is what orphaned listeners and left the
|
|
94
661
|
# EADDRINUSE race behind.
|
|
95
|
-
|
|
96
|
-
local pid=$1 sig=${2:-TERM} child
|
|
97
|
-
for child in $(
|
|
98
|
-
|
|
662
|
+
kill_frozen_tree() {
|
|
663
|
+
local pid=$1 sig=${2:-TERM} child parent unsafe=0
|
|
664
|
+
for child in $("$PGREP_BIN" -P "$pid" 2>/dev/null); do
|
|
665
|
+
kill -STOP "$child" 2>/dev/null || continue
|
|
666
|
+
parent=$("$PS_BIN" -o ppid= -p "$child" 2>/dev/null | tr -d ' ')
|
|
667
|
+
if [ "$parent" != "$pid" ]; then
|
|
668
|
+
kill -CONT "$child" 2>/dev/null || true
|
|
669
|
+
unsafe=1
|
|
670
|
+
continue
|
|
671
|
+
fi
|
|
672
|
+
kill_frozen_tree "$child" "$sig" || unsafe=1
|
|
99
673
|
done
|
|
100
674
|
kill -s "$sig" "$pid" 2>/dev/null || true
|
|
675
|
+
if [ "$sig" != "KILL" ]; then kill -CONT "$pid" 2>/dev/null || true; fi
|
|
676
|
+
[ "$unsafe" = "0" ]
|
|
677
|
+
}
|
|
678
|
+
|
|
679
|
+
kill_tree() {
|
|
680
|
+
local pid=$1 sig=${2:-TERM}
|
|
681
|
+
# Freeze before enumeration so a wrapper cannot exit and orphan its
|
|
682
|
+
# listener to PID 1 between pgrep and signal delivery.
|
|
683
|
+
kill -STOP "$pid" 2>/dev/null || return 0
|
|
684
|
+
kill_frozen_tree "$pid" "$sig"
|
|
685
|
+
}
|
|
686
|
+
|
|
687
|
+
stop_matching_identity() {
|
|
688
|
+
local pid=$1 token=$2 sig=${3:-TERM} actual
|
|
689
|
+
[ -n "$pid" ] && [ -n "$token" ] || return 2
|
|
690
|
+
# Authorization is checked while the PID is frozen. A pre-STOP check leaves
|
|
691
|
+
# a reuse window in which the signal can hit a new, unrelated process.
|
|
692
|
+
kill -STOP "$pid" 2>/dev/null || return 0
|
|
693
|
+
actual=$(process_start_token "$pid")
|
|
694
|
+
if [ -z "$actual" ] || [ "$actual" != "$token" ]; then
|
|
695
|
+
kill -CONT "$pid" 2>/dev/null || true
|
|
696
|
+
wd_log "refused signal $sig to pid $pid: frozen start identity did not match" >&2
|
|
697
|
+
return 2
|
|
698
|
+
fi
|
|
699
|
+
kill_frozen_tree "$pid" "$sig"
|
|
700
|
+
}
|
|
701
|
+
|
|
702
|
+
# Stop only the process identities captured while the old supervisor still
|
|
703
|
+
# owned them. The listener is also retained because a shell wrapper can exit
|
|
704
|
+
# and orphan its server between tree enumeration and signal delivery. Never
|
|
705
|
+
# replace this with "kill whatever owns the port" during a cutover.
|
|
706
|
+
stop_previous_owned_tree() {
|
|
707
|
+
local deadline pids foreign=0 identity_error=0
|
|
708
|
+
[ -n "$PREVIOUS_CHILD_PID" ] && [ -n "$PREVIOUS_CHILD_START" ] \
|
|
709
|
+
&& [ -n "$PREVIOUS_LISTENER_PID" ] && [ -n "$PREVIOUS_LISTENER_START" ] || {
|
|
710
|
+
wd_log "cutover ownership proof is incomplete; refusing to stop by port" >&2
|
|
711
|
+
return 1
|
|
712
|
+
}
|
|
713
|
+
stop_matching_identity "$PREVIOUS_CHILD_PID" "$PREVIOUS_CHILD_START" TERM || identity_error=1
|
|
714
|
+
# The frozen root sweep normally signals the listener too. Give its exit
|
|
715
|
+
# teardown time to drop the socket before separately touching the captured
|
|
716
|
+
# listener PID; a dying process can retain a ps row after its executable
|
|
717
|
+
# identity is already gone.
|
|
718
|
+
wd_sleep 0.5
|
|
719
|
+
if pid_is_listener "$PREVIOUS_LISTENER_PID"; then
|
|
720
|
+
stop_matching_identity "$PREVIOUS_LISTENER_PID" "$PREVIOUS_LISTENER_START" TERM || identity_error=1
|
|
721
|
+
fi
|
|
722
|
+
deadline=$(( $(date +%s) + 15 ))
|
|
723
|
+
while [ "$(date +%s)" -lt "$deadline" ]; do
|
|
724
|
+
if ! identity_matches "$PREVIOUS_CHILD_PID" "$PREVIOUS_CHILD_START" \
|
|
725
|
+
&& ! pid_is_listener "$PREVIOUS_LISTENER_PID"; then
|
|
726
|
+
break
|
|
727
|
+
fi
|
|
728
|
+
wd_sleep 0.2
|
|
729
|
+
done
|
|
730
|
+
if identity_matches "$PREVIOUS_CHILD_PID" "$PREVIOUS_CHILD_START"; then
|
|
731
|
+
stop_matching_identity "$PREVIOUS_CHILD_PID" "$PREVIOUS_CHILD_START" KILL || identity_error=1
|
|
732
|
+
fi
|
|
733
|
+
if pid_is_listener "$PREVIOUS_LISTENER_PID"; then
|
|
734
|
+
stop_matching_identity "$PREVIOUS_LISTENER_PID" "$PREVIOUS_LISTENER_START" KILL || identity_error=1
|
|
735
|
+
fi
|
|
736
|
+
wd_sleep 0.2
|
|
737
|
+
if identity_matches "$PREVIOUS_CHILD_PID" "$PREVIOUS_CHILD_START" \
|
|
738
|
+
|| pid_is_listener "$PREVIOUS_LISTENER_PID"; then
|
|
739
|
+
wd_log "captured previous child/listener identity did not exit" >&2
|
|
740
|
+
return 1
|
|
741
|
+
fi
|
|
742
|
+
pids=$(listener_pids)
|
|
743
|
+
if [ -n "$pids" ]; then
|
|
744
|
+
wd_log "port :$PORT is owned by unapproved pid(s) $(printf '%s' "$pids" | paste -sd, -); refusing arbitrary cleanup" >&2
|
|
745
|
+
foreign=1
|
|
746
|
+
fi
|
|
747
|
+
[ "$foreign" = "0" ] && [ "$identity_error" = "0" ]
|
|
748
|
+
}
|
|
749
|
+
|
|
750
|
+
kill_current_owned_attempt() {
|
|
751
|
+
local identity_error=0
|
|
752
|
+
if [ -n "${child:-}" ] && [ -n "${child_start_token:-}" ]; then
|
|
753
|
+
stop_matching_identity "$child" "$child_start_token" TERM || identity_error=1
|
|
754
|
+
fi
|
|
755
|
+
wd_sleep 0.2
|
|
756
|
+
if [ -n "${current_listener_pid:-}" ] && [ -n "${current_listener_start:-}" ] \
|
|
757
|
+
&& pid_is_listener "$current_listener_pid"; then
|
|
758
|
+
stop_matching_identity "$current_listener_pid" "$current_listener_start" TERM || identity_error=1
|
|
759
|
+
fi
|
|
760
|
+
wd_sleep 0.2
|
|
761
|
+
if [ -n "${child:-}" ] && [ -n "${child_start_token:-}" ] \
|
|
762
|
+
&& identity_matches "$child" "$child_start_token"; then
|
|
763
|
+
stop_matching_identity "$child" "$child_start_token" KILL || identity_error=1
|
|
764
|
+
fi
|
|
765
|
+
if [ -n "${current_listener_pid:-}" ] && [ -n "${current_listener_start:-}" ] \
|
|
766
|
+
&& pid_is_listener "$current_listener_pid"; then
|
|
767
|
+
stop_matching_identity "$current_listener_pid" "$current_listener_start" KILL || identity_error=1
|
|
768
|
+
fi
|
|
769
|
+
[ "$identity_error" = "0" ]
|
|
770
|
+
}
|
|
771
|
+
|
|
772
|
+
select_previous_spec() {
|
|
773
|
+
local reason=$1
|
|
774
|
+
[ -n "$PREVIOUS_START" ] && [ -n "$PREVIOUS_HOME" ] && [ -n "$PREVIOUS_REPO" ] \
|
|
775
|
+
&& [ -n "$PREVIOUS_HARNESS_ROOT" ] || return 1
|
|
776
|
+
if [ -n "$TRANSITION_PLAN_SHA256" ]; then
|
|
777
|
+
if ! guard_cmd transition-rollback "$CUTOVER_ID" --state-dir "$STATE_DIR"; then
|
|
778
|
+
wd_log "filesystem transition rollback failed — refusing to start previous over target state" >&2
|
|
779
|
+
return 1
|
|
780
|
+
fi
|
|
781
|
+
transition_rolled_back=1
|
|
782
|
+
wd_log "filesystem transition rolled back; rejected target output retained in cutover quarantine"
|
|
783
|
+
fi
|
|
784
|
+
cutover_event_required restoring "$reason"
|
|
785
|
+
START_CMD="$PREVIOUS_START"
|
|
786
|
+
DSH_ROOT="$PREVIOUS_HOME"
|
|
787
|
+
REPO="$PREVIOUS_REPO"
|
|
788
|
+
HARNESS_ROOT="$PREVIOUS_HARNESS_ROOT"
|
|
789
|
+
PROFILE="${PREVIOUS_PROFILE:-web}"
|
|
790
|
+
export DSH_HOME="$DSH_ROOT"
|
|
791
|
+
CUTOVER_ROLE="previous"
|
|
792
|
+
browser_handoff_done=0
|
|
793
|
+
browser_handoff_reported=0
|
|
794
|
+
rm -f "$BROWSER_HANDOFF_ACK_FILE"
|
|
795
|
+
failures=0
|
|
796
|
+
reset_done=1
|
|
797
|
+
port_races=0
|
|
798
|
+
rm -f "$CONTROL_FILE" "$CONTROL_ABORT_FILE" "$CONTROL_RESTORE_FILE"
|
|
799
|
+
return 0
|
|
800
|
+
}
|
|
801
|
+
|
|
802
|
+
# Return 0 when a durable operator request was consumed; control_result tells
|
|
803
|
+
# the caller whether to launch previous or park according to wait-for-user.
|
|
804
|
+
handle_cutover_control() {
|
|
805
|
+
local action effective
|
|
806
|
+
control_result=''
|
|
807
|
+
action=$(cutover_control_action)
|
|
808
|
+
[ -n "$action" ] || return 1
|
|
809
|
+
cutover_event_required control-requested "$action"
|
|
810
|
+
effective=$action
|
|
811
|
+
if [ "$action" = "abort" ]; then effective=$CUTOVER_POLICY; fi
|
|
812
|
+
if [ "$effective" = "restore-previous" ]; then
|
|
813
|
+
if ! kill_current_owned_attempt; then
|
|
814
|
+
rm -f "$CONTROL_FILE" "$CONTROL_ABORT_FILE" "$CONTROL_RESTORE_FILE"
|
|
815
|
+
cutover_event_required awaiting-user "operator requested $action, but the frozen current identity no longer matched; refusing to signal an unapproved process"
|
|
816
|
+
control_result='wait'
|
|
817
|
+
return 0
|
|
818
|
+
fi
|
|
819
|
+
if [ -n "${child:-}" ]; then wait "$child" 2>/dev/null || true; fi
|
|
820
|
+
if select_previous_spec "operator requested $action; restoring previous complete launch specification"; then
|
|
821
|
+
control_result='restore'
|
|
822
|
+
return 0
|
|
823
|
+
fi
|
|
824
|
+
effective='wait-for-user'
|
|
825
|
+
fi
|
|
826
|
+
rm -f "$CONTROL_FILE" "$CONTROL_ABORT_FILE" "$CONTROL_RESTORE_FILE"
|
|
827
|
+
cutover_event_required awaiting-user "operator requested $action; recovery is waiting for user"
|
|
828
|
+
control_result='wait'
|
|
829
|
+
return 0
|
|
830
|
+
}
|
|
831
|
+
|
|
832
|
+
# Keep a recovered/otherwise healthy child available while a non-terminal
|
|
833
|
+
# cutover waits for explicit operator action. This path intentionally skips
|
|
834
|
+
# last-good stamping and continues to consume abort/restore requests.
|
|
835
|
+
wait_cutover_with_live_child() {
|
|
836
|
+
while identity_matches "$child" "$child_start_token"; do
|
|
837
|
+
if handle_cutover_control && [ "$control_result" = "restore" ]; then return 0; fi
|
|
838
|
+
wd_sleep 2
|
|
839
|
+
done
|
|
840
|
+
wait "$child" 2>/dev/null || true
|
|
101
841
|
}
|
|
102
842
|
|
|
103
843
|
# Echo a pid and all its descendants, one per line.
|
|
104
844
|
pid_tree() {
|
|
105
845
|
local pid=$1 child
|
|
106
846
|
echo "$pid"
|
|
107
|
-
for child in $(
|
|
847
|
+
for child in $("$PGREP_BIN" -P "$pid" 2>/dev/null); do
|
|
108
848
|
pid_tree "$child"
|
|
109
849
|
done
|
|
110
850
|
}
|
|
@@ -119,7 +859,7 @@ instance_listen_ports() {
|
|
|
119
859
|
local pids
|
|
120
860
|
pids=$(pid_tree "$1" | paste -sd, -)
|
|
121
861
|
[ -n "$pids" ] || return 0
|
|
122
|
-
|
|
862
|
+
"$LSOF_BIN" -nP -a -p "$pids" -iTCP -sTCP:LISTEN 2>/dev/null \
|
|
123
863
|
| awk 'NR > 1 { n = split($9, a, ":"); print a[n] }' | sort -u
|
|
124
864
|
}
|
|
125
865
|
|
|
@@ -127,11 +867,11 @@ instance_listen_ports() {
|
|
|
127
867
|
# listener (the one-time bounce that moves a running instance under supervision).
|
|
128
868
|
free_port() {
|
|
129
869
|
local pid p
|
|
130
|
-
pid=$(
|
|
870
|
+
pid=$(listener_pids)
|
|
131
871
|
if [ -n "$pid" ]; then
|
|
132
|
-
|
|
872
|
+
wd_log "freeing :$PORT from pid(s) $pid"
|
|
133
873
|
for p in $pid; do kill_tree "$p" TERM; done
|
|
134
|
-
|
|
874
|
+
wd_sleep 2
|
|
135
875
|
fi
|
|
136
876
|
}
|
|
137
877
|
|
|
@@ -169,6 +909,60 @@ stamp_last_good_boot() {
|
|
|
169
909
|
printf '{"revision":"%s","at":%s}\n' "$sha" "$(date +%s)000" > "$STATE_DIR/last-good-boot.json"
|
|
170
910
|
}
|
|
171
911
|
|
|
912
|
+
# A healthy boot also proves the current PROFILE COMPOSITION runs: snapshot
|
|
913
|
+
# its inputs (the bundles patch layer + the profile manifest). This is the
|
|
914
|
+
# rollback target for failures a checkout reset cannot fix — a freshly
|
|
915
|
+
# installed plugin whose row breaks the real boot lives in the profile, not
|
|
916
|
+
# the repository.
|
|
917
|
+
snapshot_composition() {
|
|
918
|
+
local dir="$DSH_ROOT/profiles/$PROFILE"
|
|
919
|
+
[ -f "$dir/cordis.patch.yml" ] || return 0
|
|
920
|
+
mkdir -p "$STATE_DIR/last-good-composition"
|
|
921
|
+
cp "$dir/cordis.patch.yml" "$STATE_DIR/last-good-composition/"
|
|
922
|
+
if [ -f "$dir/package.json" ]; then cp "$dir/package.json" "$STATE_DIR/last-good-composition/"; fi
|
|
923
|
+
}
|
|
924
|
+
|
|
925
|
+
# Restore the snapshotted composition over the live one, backing the current
|
|
926
|
+
# (failing) inputs up first and computing the delta for the recovery report.
|
|
927
|
+
# Returns 1 when there is nothing to restore to (no snapshot, or the snapshot
|
|
928
|
+
# already IS the live composition — retrying a boot with unchanged inputs is
|
|
929
|
+
# pointless).
|
|
930
|
+
restore_composition() {
|
|
931
|
+
local snap="$STATE_DIR/last-good-composition"
|
|
932
|
+
local dir="$DSH_ROOT/profiles/$PROFILE"
|
|
933
|
+
[ -f "$snap/cordis.patch.yml" ] || return 1
|
|
934
|
+
local same=1
|
|
935
|
+
diff -q "$snap/cordis.patch.yml" "$dir/cordis.patch.yml" >/dev/null 2>&1 || same=0
|
|
936
|
+
if [ -f "$snap/package.json" ] || [ -f "$dir/package.json" ]; then
|
|
937
|
+
diff -q "$snap/package.json" "$dir/package.json" >/dev/null 2>&1 || same=0
|
|
938
|
+
fi
|
|
939
|
+
[ "$same" = "0" ] || return 1
|
|
940
|
+
# What the rollback unmounts — names for the recovery report.
|
|
941
|
+
comp_restore_detail=$(node -e '
|
|
942
|
+
const fs = require("fs")
|
|
943
|
+
const read = (f) => { try { return JSON.parse(fs.readFileSync(f, "utf8")) } catch { return {} } }
|
|
944
|
+
const live = read(process.argv[1]), snap = read(process.argv[2])
|
|
945
|
+
const added = (a, b) => a.filter((x) => !b.includes(x))
|
|
946
|
+
const parts = []
|
|
947
|
+
const rows = added(live.dsh?.profile?.bundles ?? [], snap.dsh?.profile?.bundles ?? [])
|
|
948
|
+
if (rows.length > 0) parts.push(`卸载挂载行: ${rows.join(", ")}`)
|
|
949
|
+
const deps = added(Object.keys(live.dependencies ?? {}), Object.keys(snap.dependencies ?? {}))
|
|
950
|
+
if (deps.length > 0) parts.push(`移除依赖: ${deps.join(", ")}`)
|
|
951
|
+
process.stdout.write(parts.join(";"))
|
|
952
|
+
' "$dir/package.json" "$snap/package.json" 2>/dev/null)
|
|
953
|
+
if ! diff -q "$snap/cordis.patch.yml" "$dir/cordis.patch.yml" >/dev/null 2>&1; then
|
|
954
|
+
comp_restore_detail="${comp_restore_detail:+$comp_restore_detail;}回滚 profile patch 层变更"
|
|
955
|
+
fi
|
|
956
|
+
local backup="$STATE_DIR/composition-backup-$(date +%s)"
|
|
957
|
+
mkdir -p "$backup"
|
|
958
|
+
cp "$dir/cordis.patch.yml" "$backup/" 2>/dev/null || true
|
|
959
|
+
if [ -f "$dir/package.json" ]; then cp "$dir/package.json" "$backup/"; fi
|
|
960
|
+
cp "$snap/cordis.patch.yml" "$dir/cordis.patch.yml"
|
|
961
|
+
if [ -f "$snap/package.json" ]; then cp "$snap/package.json" "$dir/package.json"; fi
|
|
962
|
+
wd_log "restored the last healthy profile composition over $dir (failing inputs backed up to $backup)"
|
|
963
|
+
return 0
|
|
964
|
+
}
|
|
965
|
+
|
|
172
966
|
# Roll back to a known-good revision — unless that revision already IS HEAD:
|
|
173
967
|
# the reset would be a commit no-op whose only effect is wiping uncommitted
|
|
174
968
|
# work (a real hazard with concurrent sessions on a shared checkout), so skip
|
|
@@ -178,10 +972,10 @@ rollback_to() {
|
|
|
178
972
|
local sha="$1" head
|
|
179
973
|
head=$(git -C "$REPO" rev-parse HEAD 2>/dev/null)
|
|
180
974
|
if [ -n "$head" ] && [ "$sha" = "$head" ]; then
|
|
181
|
-
|
|
975
|
+
wd_log "rollback target $sha is the current HEAD — skipping reset (nothing to roll back; a reset would only wipe uncommitted work)"
|
|
182
976
|
return 1
|
|
183
977
|
fi
|
|
184
|
-
|
|
978
|
+
wd_log "rolling repo back to last known-good $sha"
|
|
185
979
|
guard_reset "$sha"
|
|
186
980
|
}
|
|
187
981
|
|
|
@@ -213,7 +1007,10 @@ guard_cmd() {
|
|
|
213
1007
|
}
|
|
214
1008
|
|
|
215
1009
|
guard_verify() {
|
|
216
|
-
|
|
1010
|
+
# Revalidate the exact short-lived authorization selected before the stop.
|
|
1011
|
+
# A same-launch proof may outlive the original build credential's freshness,
|
|
1012
|
+
# but only while every fingerprinted deployment input remains identical.
|
|
1013
|
+
guard_cmd verify-restart --repo "$REPO" --state-dir "$STATE_DIR" >/dev/null 2>&1
|
|
217
1014
|
}
|
|
218
1015
|
|
|
219
1016
|
guard_reset() {
|
|
@@ -224,10 +1021,15 @@ guard_reset() {
|
|
|
224
1021
|
# retry button that signals the watchdog (SIGUSR1) — no terminal needed.
|
|
225
1022
|
page_script() {
|
|
226
1023
|
cat <<'EOF'
|
|
1024
|
+
if (process.env.ANKH_GUARD_TEST_RUN_DIR && process.env.ANKH_GUARD_TEST_RUN_TOKEN
|
|
1025
|
+
&& process.env.ANKH_GUARD_TEST_REGISTER_BIN) {
|
|
1026
|
+
require('child_process').spawnSync(process.execPath, [process.env.ANKH_GUARD_TEST_REGISTER_BIN,
|
|
1027
|
+
'register-pid', String(process.pid), 'crash-page'], { stdio: 'ignore', env: process.env });
|
|
1028
|
+
}
|
|
227
1029
|
const http = require('http');
|
|
228
1030
|
const port = Number(process.env.WD_PORT || 3080);
|
|
229
1031
|
const wd = Number(process.env.WD_PID);
|
|
230
|
-
|
|
1032
|
+
const handler = (req, res) => {
|
|
231
1033
|
if (req.url === '/restart') {
|
|
232
1034
|
try { process.kill(wd, 'SIGUSR1'); res.end('retrying...'); }
|
|
233
1035
|
catch (e) { res.statusCode = 500; res.end('signal failed: ' + e.message); }
|
|
@@ -242,12 +1044,46 @@ http.createServer((req, res) => {
|
|
|
242
1044
|
+ '<div style="text-align:center"><h2>dsh 服务未能启动</h2>'
|
|
243
1045
|
+ '<p>看门狗多次尝试仍未拉起服务。点击重试,或查看看门狗日志。</p>'
|
|
244
1046
|
+ '<form action="/restart"><button style="font-size:18px;padding:10px 28px">重试</button></form></div></body>');
|
|
245
|
-
}
|
|
1047
|
+
};
|
|
1048
|
+
// An occupied port is the COMMON case at give-up (the boot failures were
|
|
1049
|
+
// often EADDRINUSE themselves). Dying on the bind error — an unhandled
|
|
1050
|
+
// 'error' event — would return the watchdog's `wait` and drop it back into
|
|
1051
|
+
// the boot loop, fighting the healthy occupant it just gave up against
|
|
1052
|
+
// (observed 2026-08-30 in an e2e rig: four give-up cycles, the occupant
|
|
1053
|
+
// killed over and over). Park instead: retry the bind every 5s; SIGUSR1
|
|
1054
|
+
// still re-arms the boot loop.
|
|
1055
|
+
let noted = false;
|
|
1056
|
+
function bind() {
|
|
1057
|
+
const server = http.createServer(handler);
|
|
1058
|
+
server.on('error', (e) => {
|
|
1059
|
+
if (e && e.code === 'EADDRINUSE') {
|
|
1060
|
+
if (!noted) {
|
|
1061
|
+
noted = true;
|
|
1062
|
+
process.stderr.write(new Date().toISOString() + ' [watchdog] crash page cannot bind :' + port + ' (occupied) — retrying every 5s; SIGUSR1 to ' + wd + ' re-arms the boot loop\n');
|
|
1063
|
+
}
|
|
1064
|
+
setTimeout(bind, 5000);
|
|
1065
|
+
return;
|
|
1066
|
+
}
|
|
1067
|
+
throw e;
|
|
1068
|
+
});
|
|
1069
|
+
server.listen(port, '127.0.0.1');
|
|
1070
|
+
}
|
|
1071
|
+
|
|
1072
|
+
bind();
|
|
246
1073
|
EOF
|
|
247
1074
|
}
|
|
248
1075
|
|
|
1076
|
+
park_cutover() {
|
|
1077
|
+
local reason=$1
|
|
1078
|
+
printf '%s launch cutover waiting: %s\n' "$(date '+%F %T')" "$reason" > "$GIVE_UP_MARKER"
|
|
1079
|
+
WD_PORT="$PORT" WD_PID="$$" node -e "$(page_script)" &
|
|
1080
|
+
page_pid=$!
|
|
1081
|
+
wait "$page_pid" 2>/dev/null || true
|
|
1082
|
+
page_pid=''
|
|
1083
|
+
}
|
|
1084
|
+
|
|
249
1085
|
retry_on_usrs() {
|
|
250
|
-
|
|
1086
|
+
wd_log "USR1 received — clearing give-up marker and retrying"
|
|
251
1087
|
rm -f "$GIVE_UP_MARKER"
|
|
252
1088
|
if [ -n "${page_pid:-}" ]; then kill "$page_pid" 2>/dev/null; fi
|
|
253
1089
|
failures=0
|
|
@@ -255,6 +1091,10 @@ retry_on_usrs() {
|
|
|
255
1091
|
port_races=0
|
|
256
1092
|
}
|
|
257
1093
|
|
|
1094
|
+
# Register before the first ownership branch can exit or detach more children.
|
|
1095
|
+
test_register_self "${ANKH_GUARD_TEST_PROCESS_ROLE:-watchdog}"
|
|
1096
|
+
test_event_self "${ANKH_GUARD_TEST_PROCESS_ROLE:-watchdog}" process-started
|
|
1097
|
+
|
|
258
1098
|
# --supervise: one watchdog only. The claim must be atomic — a check-then-write
|
|
259
1099
|
# (`[ -f ]` + `kill -0`, then `>`) is a TOCTOU window in which two watchdogs
|
|
260
1100
|
# starting together both find no live owner, both write, and both supervise the
|
|
@@ -266,25 +1106,69 @@ if [ "$SUPERVISE" = "1" ]; then
|
|
|
266
1106
|
mkdir -p "$(dirname "$PIDFILE")" 2>/dev/null || true
|
|
267
1107
|
claimed=0
|
|
268
1108
|
attempt=0
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
if (set -C; echo $$ > "$PIDFILE") 2>/dev/null; then claimed=1; break; fi
|
|
1109
|
+
empty_reads=0
|
|
1110
|
+
if [ -n "${WD_TAKEOVER_FROM:-}" ]; then
|
|
272
1111
|
owner=$(cat "$PIDFILE" 2>/dev/null)
|
|
273
|
-
if [ -
|
|
274
|
-
|
|
275
|
-
|
|
1112
|
+
if [ -z "${WD_TAKEOVER_FROM_START:-}" ] || [ "$owner" != "$WD_TAKEOVER_FROM" ] \
|
|
1113
|
+
|| ! identity_matches "$WD_TAKEOVER_FROM" "$WD_TAKEOVER_FROM_START"; then
|
|
1114
|
+
wd_log "takeover refused: expected matching pidfile owner identity $WD_TAKEOVER_FROM, found ${owner:-none}" >&2
|
|
1115
|
+
exit 1
|
|
276
1116
|
fi
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
#
|
|
280
|
-
|
|
281
|
-
|
|
1117
|
+
takeover_tmp="$PIDFILE.takeover.$$"
|
|
1118
|
+
echo $$ > "$takeover_tmp"
|
|
1119
|
+
# Atomic rename is the ownership handoff commit point. The previous
|
|
1120
|
+
# watchdog already knows how to yield to a different live pidfile owner
|
|
1121
|
+
# without reaping its child, so this works even when that watchdog is the
|
|
1122
|
+
# older package version that had no reconfigure verb.
|
|
1123
|
+
mv -f "$takeover_tmp" "$PIDFILE"
|
|
1124
|
+
claimed=1
|
|
1125
|
+
wd_log "claimed supervision from watchdog $owner; old child remains running until the scheduled exit"
|
|
1126
|
+
supervisor_start_token=$(process_start_token "$$")
|
|
1127
|
+
[ -n "$supervisor_start_token" ] || { wd_log "could not capture replacement watchdog start identity" >&2; exit 1; }
|
|
1128
|
+
cutover_event_required supervisor-ready "$$" "$supervisor_start_token"
|
|
1129
|
+
else
|
|
1130
|
+
while [ "$attempt" -lt 5 ]; do
|
|
1131
|
+
attempt=$((attempt + 1))
|
|
1132
|
+
if (set -C; echo $$ > "$PIDFILE") 2>/dev/null; then claimed=1; break; fi
|
|
1133
|
+
owner=$(cat "$PIDFILE" 2>/dev/null)
|
|
1134
|
+
if [ -n "$owner" ]; then
|
|
1135
|
+
if kill -0 "$owner" 2>/dev/null; then
|
|
1136
|
+
wd_log "already supervised by pid $owner; exiting"
|
|
1137
|
+
exit 0
|
|
1138
|
+
fi
|
|
1139
|
+
# A real but dead claim: safe to drop below.
|
|
1140
|
+
empty_reads=0
|
|
1141
|
+
else
|
|
1142
|
+
# An EMPTY pidfile is a rival's claim mid-write: the noclobber create
|
|
1143
|
+
# and the echo are two disk operations, and a preempted winner sits
|
|
1144
|
+
# between them. Deleting the file here re-opens the race and can
|
|
1145
|
+
# cascade until every racer exhausts its attempts (observed under
|
|
1146
|
+
# deploy-gate load: 8 concurrent racers, zero survivors). Give the
|
|
1147
|
+
# writer a beat to land its pid; only treat the file as abandoned
|
|
1148
|
+
# after several consecutive empty reads.
|
|
1149
|
+
empty_reads=$((empty_reads + 1))
|
|
1150
|
+
if [ "$empty_reads" -le 3 ]; then
|
|
1151
|
+
attempt=$((attempt - 1))
|
|
1152
|
+
wd_sleep 0.2
|
|
1153
|
+
continue
|
|
1154
|
+
fi
|
|
1155
|
+
fi
|
|
1156
|
+
# Stale (owner gone, or abandoned mid-write): drop it and race for the
|
|
1157
|
+
# claim again. Losing that race is correct — the next pass sees a live
|
|
1158
|
+
# owner and exits through the branch above.
|
|
1159
|
+
rm -f "$PIDFILE"
|
|
1160
|
+
done
|
|
1161
|
+
fi
|
|
282
1162
|
if [ "$claimed" != "1" ]; then
|
|
283
|
-
|
|
1163
|
+
wd_log "could not claim $PIDFILE after $attempt attempts" >&2
|
|
284
1164
|
exit 1
|
|
285
1165
|
fi
|
|
286
1166
|
fi
|
|
287
1167
|
|
|
1168
|
+
if [ "$SUPERVISE" = "1" ] && [ "$claimed" = "1" ]; then
|
|
1169
|
+
test_event_self "${ANKH_GUARD_TEST_PROCESS_ROLE:-watchdog}" pidfile-published
|
|
1170
|
+
fi
|
|
1171
|
+
|
|
288
1172
|
# Graceful launch: let the scheduling turn finish before the adoption bounce.
|
|
289
1173
|
if [ "$DELAY" -gt 0 ] 2>/dev/null; then sleep "$DELAY"; fi
|
|
290
1174
|
|
|
@@ -292,7 +1176,14 @@ rm -f "$GIVE_UP_MARKER"
|
|
|
292
1176
|
failures=0
|
|
293
1177
|
reset_done=0
|
|
294
1178
|
port_races=0
|
|
1179
|
+
target_attempt=0
|
|
1180
|
+
previous_attempt=0
|
|
295
1181
|
yielded=0
|
|
1182
|
+
comp_restore_done=0
|
|
1183
|
+
comp_restored=0
|
|
1184
|
+
comp_restore_detail=''
|
|
1185
|
+
transition_applied=0
|
|
1186
|
+
transition_rolled_back=0
|
|
296
1187
|
|
|
297
1188
|
trap 'retry_on_usrs' USR1
|
|
298
1189
|
|
|
@@ -302,43 +1193,228 @@ trap 'retry_on_usrs' USR1
|
|
|
302
1193
|
# EADDRINUSE branch then had to free. SIGKILL cannot be trapped; the next
|
|
303
1194
|
# start's free_port covers that case. (`set -u` — guard every var.)
|
|
304
1195
|
cleanup() {
|
|
1196
|
+
if [ -n "${handoff_cookie_jar:-}" ]; then rm -f "$handoff_cookie_jar"; fi
|
|
305
1197
|
if [ -n "${page_pid:-}" ]; then kill "$page_pid" 2>/dev/null; fi
|
|
306
1198
|
# A YIELDING watchdog leaves its instance running for the new owner (the
|
|
307
1199
|
# port is healthy; killing it would just make the successor respawn).
|
|
308
|
-
if [ -n "${child:-}" ] && [ "${yielded:-0}" != "1" ]; then
|
|
1200
|
+
if [ -n "${child:-}" ] && [ "${yielded:-0}" != "1" ]; then
|
|
1201
|
+
if [ -n "${CUTOVER_ID:-}" ] && [ -n "${child_start_token:-}" ]; then
|
|
1202
|
+
stop_matching_identity "$child" "$child_start_token" TERM || true
|
|
1203
|
+
else
|
|
1204
|
+
kill_tree "$child" TERM
|
|
1205
|
+
fi
|
|
1206
|
+
fi
|
|
309
1207
|
# Drop the pidfile ONLY while it names us: a successor watchdog may have
|
|
310
1208
|
# already claimed it in the restart window, and deleting theirs would let a
|
|
311
1209
|
# second supervisor in.
|
|
312
1210
|
if [ -f "$PIDFILE" ] && [ "$(cat "$PIDFILE" 2>/dev/null)" = "$$" ]; then
|
|
313
|
-
|
|
1211
|
+
if [ -n "${WD_TAKEOVER_FROM:-}" ] && [ -n "${WD_TAKEOVER_FROM_START:-}" ] \
|
|
1212
|
+
&& identity_matches "$WD_TAKEOVER_FROM" "$WD_TAKEOVER_FROM_START"; then
|
|
1213
|
+
takeover_restore="$PIDFILE.restore.$$"
|
|
1214
|
+
echo "$WD_TAKEOVER_FROM" > "$takeover_restore"
|
|
1215
|
+
mv -f "$takeover_restore" "$PIDFILE"
|
|
1216
|
+
wd_log "takeover aborted while old watchdog $WD_TAKEOVER_FROM is alive — restored its pidfile claim"
|
|
1217
|
+
else
|
|
1218
|
+
rm -f "$PIDFILE"
|
|
1219
|
+
fi
|
|
314
1220
|
fi
|
|
315
1221
|
return 0
|
|
316
1222
|
}
|
|
317
1223
|
trap cleanup EXIT
|
|
318
1224
|
trap 'cleanup; exit 143' TERM INT
|
|
319
1225
|
|
|
320
|
-
|
|
1226
|
+
cutover_control_signal() {
|
|
1227
|
+
# The durable marker is the signal. Let the main loop freeze and reap the
|
|
1228
|
+
# authoritative tree; killing only the wrapper here recreates the orphan
|
|
1229
|
+
# listener race.
|
|
1230
|
+
if [ -n "${page_pid:-}" ]; then kill "$page_pid" 2>/dev/null || true; fi
|
|
1231
|
+
}
|
|
1232
|
+
trap 'cutover_control_signal' USR2
|
|
1233
|
+
|
|
1234
|
+
# Test readiness is stronger than pidfile publication: the control handler is
|
|
1235
|
+
# installed and the shell has yielded through one scheduler tick. Production
|
|
1236
|
+
# readiness remains unchanged; only explicit test event consumers observe it.
|
|
1237
|
+
test_event_self "${ANKH_GUARD_TEST_PROCESS_ROLE:-watchdog}" handler-installed
|
|
1238
|
+
if [ -n "${ANKH_GUARD_TEST_RUN_DIR:-}" ]; then wd_sleep 0.01; fi
|
|
1239
|
+
test_event_self "${ANKH_GUARD_TEST_PROCESS_ROLE:-watchdog}" keepalive-first-tick
|
|
1240
|
+
test_event_self "${ANKH_GUARD_TEST_PROCESS_ROLE:-watchdog}" ready
|
|
1241
|
+
|
|
1242
|
+
write_cutover_restart_marker() {
|
|
1243
|
+
node -e '
|
|
1244
|
+
const fs = require("fs")
|
|
1245
|
+
const [file, id, initiator] = process.argv.slice(1)
|
|
1246
|
+
fs.writeFileSync(file, JSON.stringify({
|
|
1247
|
+
reason: "launch configuration cutover",
|
|
1248
|
+
cutoverId: id,
|
|
1249
|
+
requestedAt: Date.now(),
|
|
1250
|
+
...(initiator === "" ? {} : { initiator }),
|
|
1251
|
+
}) + "\n")
|
|
1252
|
+
' "$RESTART_MARKER" "$CUTOVER_ID" "${WD_INITIATOR:-}"
|
|
1253
|
+
}
|
|
1254
|
+
|
|
1255
|
+
consume_takeover_control() {
|
|
1256
|
+
if handle_cutover_control; then
|
|
1257
|
+
if [ "$control_result" = "wait" ]; then
|
|
1258
|
+
wd_log "operator control parked cutover before the irreversible boundary; previous host remains available"
|
|
1259
|
+
park_cutover "operator abort follows wait-for-user policy"
|
|
1260
|
+
rm -f "$GIVE_UP_MARKER"
|
|
1261
|
+
else
|
|
1262
|
+
wd_log "operator control selected previous launch configuration before takeover"
|
|
1263
|
+
fi
|
|
1264
|
+
fi
|
|
1265
|
+
}
|
|
1266
|
+
|
|
1267
|
+
if [ -n "${WD_TAKEOVER_FROM:-}" ]; then
|
|
1268
|
+
# The atomic pidfile claim above is the cutover commit point. From here the
|
|
1269
|
+
# replacement watchdog — not the short-lived reconfigure caller — owns the
|
|
1270
|
+
# delayed old-child stop, so a caller/session death cannot strand the
|
|
1271
|
+
# transaction between "supervisor-ready" and "host stopped".
|
|
1272
|
+
cutover_delay="${WD_CUTOVER_DELAY_SECONDS:-5}"
|
|
1273
|
+
wd_log "supervision claimed; leaving the old host uninterrupted for ${cutover_delay}s"
|
|
1274
|
+
cutover_delay_deadline=$(node -e '
|
|
1275
|
+
const delay = Number(process.argv[1])
|
|
1276
|
+
if (!Number.isFinite(delay) || delay < 0) process.exit(1)
|
|
1277
|
+
process.stdout.write(String(Date.now() + delay * 1000))
|
|
1278
|
+
' "$cutover_delay") || { wd_log "invalid WD_CUTOVER_DELAY_SECONDS" >&2; exit 1; }
|
|
1279
|
+
while [ "$(now_ms)" -lt "$cutover_delay_deadline" ]; do
|
|
1280
|
+
consume_takeover_control
|
|
1281
|
+
wd_sleep 0.2
|
|
1282
|
+
done
|
|
1283
|
+
# Do not publish the restart marker while the previous watchdog is still
|
|
1284
|
+
# alive. Older watchdogs consume that marker themselves; if one wins that
|
|
1285
|
+
# race, the replacement child can become healthy while the cutover receipt
|
|
1286
|
+
# remains permanently nonterminal. The previous child stays up throughout
|
|
1287
|
+
# this wait. A healthy old watchdog notices our pidfile claim on its next
|
|
1288
|
+
# supervision pass and yields without reaping the child.
|
|
1289
|
+
supervisor_yield_timeout_ms="${WD_SUPERVISOR_YIELD_TIMEOUT_MS:-15000}"
|
|
1290
|
+
case "$supervisor_yield_timeout_ms" in ''|*[!0-9]*) wd_log "invalid WD_SUPERVISOR_YIELD_TIMEOUT_MS" >&2; exit 1 ;; esac
|
|
1291
|
+
[ "$supervisor_yield_timeout_ms" -ge 100 ] || { wd_log "WD_SUPERVISOR_YIELD_TIMEOUT_MS must be at least 100" >&2; exit 1; }
|
|
1292
|
+
supervisor_yield_deadline=$(( $(now_ms) + supervisor_yield_timeout_ms ))
|
|
1293
|
+
supervisor_was_live=0
|
|
1294
|
+
supervisor_timed_out=0
|
|
1295
|
+
supervisor_retirement='identity-gone'
|
|
1296
|
+
if identity_matches "$WD_TAKEOVER_FROM" "$WD_TAKEOVER_FROM_START"; then
|
|
1297
|
+
supervisor_was_live=1
|
|
1298
|
+
wd_log "waiting up to ${supervisor_yield_timeout_ms}ms for old watchdog $WD_TAKEOVER_FROM to yield; old host remains available"
|
|
1299
|
+
fi
|
|
1300
|
+
while identity_matches "$WD_TAKEOVER_FROM" "$WD_TAKEOVER_FROM_START"; do
|
|
1301
|
+
consume_takeover_control
|
|
1302
|
+
if [ "$(now_ms)" -ge "$supervisor_yield_deadline" ]; then
|
|
1303
|
+
supervisor_timed_out=1
|
|
1304
|
+
wd_log "old watchdog $WD_TAKEOVER_FROM did not yield in ${supervisor_yield_timeout_ms}ms; retiring its frozen, revalidated identity"
|
|
1305
|
+
if stop_matching_identity "$WD_TAKEOVER_FROM" "$WD_TAKEOVER_FROM_START" TERM; then
|
|
1306
|
+
supervisor_retirement='forced'
|
|
1307
|
+
retirement_deadline=$(( $(date +%s) + 3 ))
|
|
1308
|
+
while identity_matches "$WD_TAKEOVER_FROM" "$WD_TAKEOVER_FROM_START" \
|
|
1309
|
+
&& [ "$(date +%s)" -lt "$retirement_deadline" ]; do wd_sleep 0.2; done
|
|
1310
|
+
stop_matching_identity "$WD_TAKEOVER_FROM" "$WD_TAKEOVER_FROM_START" KILL || true
|
|
1311
|
+
else
|
|
1312
|
+
# The PID changed between the loop predicate and SIGSTOP. The helper
|
|
1313
|
+
# resumed it without delivering TERM; the authorized old identity is
|
|
1314
|
+
# gone, so proceed using only the separately captured child identities.
|
|
1315
|
+
supervisor_retirement='identity-gone'
|
|
1316
|
+
fi
|
|
1317
|
+
break
|
|
1318
|
+
fi
|
|
1319
|
+
wd_sleep 0.2
|
|
1320
|
+
done
|
|
1321
|
+
if [ "$supervisor_was_live" = "1" ] && [ "$supervisor_timed_out" = "0" ]; then
|
|
1322
|
+
supervisor_retirement='yielded'
|
|
1323
|
+
fi
|
|
1324
|
+
cutover_event_required previous-supervisor-retired "$supervisor_retirement"
|
|
1325
|
+
# Publish the intentional-restart marker only at the irreversible boundary.
|
|
1326
|
+
# Writing it in the reconfigure caller lets the OLD watchdog consume and
|
|
1327
|
+
# clear it before yielding, leaving the final child ready but the receipt
|
|
1328
|
+
# permanently nonterminal. The old instance still sees the marker during
|
|
1329
|
+
# SIGTERM and can snapshot interrupted sessions with the correct initiator.
|
|
1330
|
+
write_cutover_restart_marker
|
|
1331
|
+
if ! stop_previous_owned_tree; then
|
|
1332
|
+
WD_TAKEOVER_FROM=""
|
|
1333
|
+
cutover_event_required awaiting-user "could not stop the captured previous child/listener identity without touching an unapproved port owner"
|
|
1334
|
+
printf '%s launch cutover waiting: previous ownership could not be retired\n' "$(date '+%F %T')" > "$GIVE_UP_MARKER"
|
|
1335
|
+
WD_PORT="$PORT" WD_PID="$$" node -e "$(page_script)" &
|
|
1336
|
+
page_pid=$!
|
|
1337
|
+
wait "$page_pid"
|
|
1338
|
+
exit 1
|
|
1339
|
+
fi
|
|
1340
|
+
wd_log "captured previous child $PREVIOUS_CHILD_PID and listener $PREVIOUS_LISTENER_PID exited — taking over :$PORT"
|
|
1341
|
+
# This is the irreversible boundary. Never restore a possibly recycled old
|
|
1342
|
+
# supervisor pid during a much later cleanup.
|
|
1343
|
+
WD_TAKEOVER_FROM=""
|
|
1344
|
+
elif [ -n "$CUTOVER_ID" ]; then
|
|
1345
|
+
# OS-level crash recovery: this watchdog claimed a stale/empty pidfile and
|
|
1346
|
+
# resumes the atomically selected side of an existing transaction. Replace
|
|
1347
|
+
# any orphan listener from the failed supervisor, then prove a fresh final
|
|
1348
|
+
# child; never compact the transaction merely because its driver died.
|
|
1349
|
+
wd_log "resuming launch cutover $CUTOVER_ID on selected side $CUTOVER_ROLE"
|
|
1350
|
+
supervisor_start_token=$(process_start_token "$$")
|
|
1351
|
+
[ -n "$supervisor_start_token" ] || { wd_log "could not capture resumed watchdog start identity" >&2; exit 1; }
|
|
1352
|
+
cutover_event_required supervisor-ready "$$" "$supervisor_start_token"
|
|
1353
|
+
write_cutover_restart_marker
|
|
1354
|
+
if ! stop_previous_owned_tree; then
|
|
1355
|
+
cutover_event_required awaiting-user "resume could not prove the shared port free from the captured previous identity"
|
|
1356
|
+
printf '%s launch cutover waiting: port ownership is ambiguous\n' "$(date '+%F %T')" > "$GIVE_UP_MARKER"
|
|
1357
|
+
WD_PORT="$PORT" WD_PID="$$" node -e "$(page_script)" &
|
|
1358
|
+
page_pid=$!
|
|
1359
|
+
wait "$page_pid"
|
|
1360
|
+
exit 1
|
|
1361
|
+
fi
|
|
1362
|
+
elif [ "${WD_WAIT_OWNER:-0}" = "1" ]; then
|
|
321
1363
|
# Adoption ahead of a self-restart: the current owner exits on its own.
|
|
322
|
-
|
|
323
|
-
while
|
|
324
|
-
|
|
1364
|
+
wd_log "waiting for the current owner of :$PORT to exit"
|
|
1365
|
+
while "$LSOF_BIN" -tiTCP:"$PORT" -sTCP:LISTEN -P >/dev/null 2>&1; do wd_sleep 1; done
|
|
1366
|
+
wd_log "port free — taking over"
|
|
325
1367
|
else
|
|
326
1368
|
free_port
|
|
327
1369
|
fi
|
|
328
1370
|
|
|
329
1371
|
while true; do
|
|
330
|
-
# Self-heal the ownership claim
|
|
331
|
-
#
|
|
332
|
-
#
|
|
333
|
-
# instance (observed: stale watchdog + deleted pidfile → second watchdog
|
|
334
|
-
# spawned → both fought over the port).
|
|
1372
|
+
# Self-heal the ownership claim before processing control or changing live
|
|
1373
|
+
# state. If another supervisor owns the pidfile, this process has no
|
|
1374
|
+
# authority to apply or roll back a filesystem transition.
|
|
335
1375
|
if [ "$SUPERVISE" = "1" ]; then
|
|
336
1376
|
if [ ! -f "$PIDFILE" ]; then (set -C; echo $$ > "$PIDFILE") 2>/dev/null || true; fi
|
|
337
1377
|
pidowner=$(cat "$PIDFILE" 2>/dev/null)
|
|
338
1378
|
if [ -n "$pidowner" ] && [ "$pidowner" != "$$" ] && kill -0 "$pidowner" 2>/dev/null; then
|
|
339
|
-
|
|
1379
|
+
wd_log "pidfile now owned by live pid $pidowner — yielding"
|
|
340
1380
|
yielded=1
|
|
341
|
-
|
|
1381
|
+
# Non-zero keeps launchd/systemd's stable launcher alive: it restarts,
|
|
1382
|
+
# reads the newly selected durable spec, then waits behind the successor.
|
|
1383
|
+
# A detached parent simply observes the code and is unaffected.
|
|
1384
|
+
exit 75
|
|
1385
|
+
fi
|
|
1386
|
+
fi
|
|
1387
|
+
if [ -n "$CUTOVER_ID" ] && handle_cutover_control; then
|
|
1388
|
+
if [ "$control_result" = "wait" ]; then
|
|
1389
|
+
park_cutover "operator abort follows wait-for-user policy"
|
|
1390
|
+
continue
|
|
1391
|
+
fi
|
|
1392
|
+
wd_log "operator control selected the previous complete launch specification"
|
|
1393
|
+
fi
|
|
1394
|
+
if [ -n "$TRANSITION_PLAN_SHA256" ]; then
|
|
1395
|
+
if [ "$CUTOVER_ROLE" = "target" ] && [ "$transition_applied" = "0" ]; then
|
|
1396
|
+
if guard_cmd transition-apply "$CUTOVER_ID" --state-dir "$STATE_DIR"; then
|
|
1397
|
+
transition_applied=1
|
|
1398
|
+
wd_log "filesystem transition applied after previous stopped and before target start"
|
|
1399
|
+
else
|
|
1400
|
+
wd_log "filesystem transition apply failed — target will not start" >&2
|
|
1401
|
+
if [ "$CUTOVER_POLICY" = "restore-previous" ] \
|
|
1402
|
+
&& select_previous_spec "filesystem transition failed before target start; restoring previous spec"; then
|
|
1403
|
+
continue
|
|
1404
|
+
fi
|
|
1405
|
+
cutover_event_required awaiting-user "filesystem transition failed; target was not started and previous was not restored"
|
|
1406
|
+
park_cutover "filesystem transition failed; inspect the cutover transition journal"
|
|
1407
|
+
continue
|
|
1408
|
+
fi
|
|
1409
|
+
elif [ "$CUTOVER_ROLE" = "previous" ] && [ "$transition_rolled_back" = "0" ]; then
|
|
1410
|
+
if guard_cmd transition-rollback "$CUTOVER_ID" --state-dir "$STATE_DIR"; then
|
|
1411
|
+
transition_rolled_back=1
|
|
1412
|
+
wd_log "filesystem transition rollback verified before previous start"
|
|
1413
|
+
else
|
|
1414
|
+
cutover_event_required awaiting-user "filesystem transition rollback failed; previous was not started"
|
|
1415
|
+
park_cutover "filesystem transition rollback failed; previous remains stopped"
|
|
1416
|
+
continue
|
|
1417
|
+
fi
|
|
342
1418
|
fi
|
|
343
1419
|
fi
|
|
344
1420
|
# Snapshot BEFORE this boot rewrites it: the stamp exists iff this
|
|
@@ -348,7 +1424,16 @@ while true; do
|
|
|
348
1424
|
# watchdog still judges correctly.
|
|
349
1425
|
had_boot_stamp=0
|
|
350
1426
|
[ -f "$STATE_DIR/last-good-boot.json" ] && had_boot_stamp=1
|
|
351
|
-
|
|
1427
|
+
if [ "$CUTOVER_ROLE" = "target" ]; then
|
|
1428
|
+
target_attempt=$((target_attempt + 1))
|
|
1429
|
+
current_attempt=$target_attempt
|
|
1430
|
+
else
|
|
1431
|
+
previous_attempt=$((previous_attempt + 1))
|
|
1432
|
+
current_attempt=$previous_attempt
|
|
1433
|
+
fi
|
|
1434
|
+
# Keep the long-standing "starting instance" prefix stable for operators and
|
|
1435
|
+
# log consumers; the role is additive cutover metadata.
|
|
1436
|
+
wd_log "starting instance on :$PORT (role=$CUTOVER_ROLE, failures=$failures, attempt=$current_attempt)"
|
|
352
1437
|
# Capture this attempt's output for failure-domain classification. Plain
|
|
353
1438
|
# redirection only — never > >(tee …) process substitution: a sandboxed or
|
|
354
1439
|
# detached spawner can EPERM on the /dev/fd/N that >() opens (workspace-write
|
|
@@ -356,32 +1441,91 @@ while true; do
|
|
|
356
1441
|
# log is mirrored into this log below; a healthy run's boot message names
|
|
357
1442
|
# the file its output lives in.
|
|
358
1443
|
: > "$ATTEMPT_LOG"
|
|
1444
|
+
chmod 600 "$ATTEMPT_LOG" 2>/dev/null || true
|
|
359
1445
|
launch_instance > "$ATTEMPT_LOG" 2>&1 &
|
|
360
1446
|
child=$!
|
|
361
|
-
|
|
1447
|
+
child_start_token=$(process_start_token "$child")
|
|
1448
|
+
[ -n "$child_start_token" ] || child_start_token="unavailable-$child"
|
|
1449
|
+
cutover_event_required child-started "$CUTOVER_ROLE" "$current_attempt" "$child" "$child_start_token"
|
|
1450
|
+
last_transport_status=''
|
|
1451
|
+
launch_url_reported=0
|
|
1452
|
+
launch_url_value=''
|
|
1453
|
+
readiness_detail=''
|
|
1454
|
+
protected_ready=0
|
|
1455
|
+
# Launch URLs and browser cookies are process-bound. Every retry must prove
|
|
1456
|
+
# and hand off its own URL; an accepted URL from a rejected child is stale.
|
|
1457
|
+
browser_handoff_done=0
|
|
1458
|
+
browser_handoff_reported=0
|
|
1459
|
+
current_listener_pid=''
|
|
1460
|
+
current_listener_start=''
|
|
1461
|
+
# Boot window: transport-up is not enough. A protected root can answer 401;
|
|
1462
|
+
# ready_probe completes the process's announced launch-URL cookie exchange.
|
|
362
1463
|
up=0
|
|
1464
|
+
readiness_failure_detail="readiness not proven within ${BOOT_TIMEOUT}s"
|
|
363
1465
|
boot_limit=$(( $(date +%s) + BOOT_TIMEOUT ))
|
|
364
1466
|
while [ "$(date +%s)" -lt "$boot_limit" ]; do
|
|
365
|
-
if
|
|
366
|
-
|
|
367
|
-
|
|
1467
|
+
if [ -n "$(cutover_control_action)" ]; then
|
|
1468
|
+
readiness_failure_detail="operator control interrupted readiness"
|
|
1469
|
+
kill_current_owned_attempt
|
|
1470
|
+
break
|
|
1471
|
+
fi
|
|
1472
|
+
if ! kill -0 "$child" 2>/dev/null; then
|
|
1473
|
+
readiness_failure_detail="child exited before ownership-stable readiness"
|
|
1474
|
+
break
|
|
1475
|
+
fi
|
|
1476
|
+
if ready_probe; then
|
|
1477
|
+
if prove_stable_readiness; then
|
|
1478
|
+
up=1
|
|
1479
|
+
else
|
|
1480
|
+
readiness_failure_detail="provisional readiness did not retain one child/listener identity through the stability window"
|
|
1481
|
+
fi
|
|
1482
|
+
# Readiness that cannot hold the same child/listener identity for the
|
|
1483
|
+
# stability window is an attempt failure, not an invitation to attach
|
|
1484
|
+
# to whichever process next answers on the shared port.
|
|
1485
|
+
break
|
|
1486
|
+
fi
|
|
1487
|
+
wd_sleep 1
|
|
368
1488
|
done
|
|
369
1489
|
|
|
370
1490
|
if [ "$up" = "0" ]; then
|
|
371
1491
|
# Read the bound ports BEFORE reaping — once the child is gone there is no
|
|
372
1492
|
# way left to tell "never started" from "started on the wrong port".
|
|
373
1493
|
bound=""
|
|
374
|
-
if
|
|
1494
|
+
if identity_matches "$child" "$child_start_token"; then bound=$(instance_listen_ports "$child"); fi
|
|
375
1495
|
# Never came up (or died); stop a still-alive child and reap it.
|
|
376
|
-
|
|
1496
|
+
kill_current_owned_attempt
|
|
377
1497
|
wait "$child" 2>/dev/null
|
|
1498
|
+
# Strip bearer launch URLs before any durable failure output is mirrored.
|
|
1499
|
+
redact_launch_urls_in_output
|
|
378
1500
|
# Mirror the captured output into the watchdog log: with plain redirection
|
|
379
1501
|
# (see the launch site) the attempt log is the only place the failure was
|
|
380
1502
|
# written, and the watchdog log is where an operator looks first.
|
|
381
1503
|
sed 's/^/[instance] /' "$ATTEMPT_LOG" 2>/dev/null
|
|
382
1504
|
|
|
1505
|
+
if [ -n "$CUTOVER_ID" ] && [ -n "$(cutover_control_action)" ]; then
|
|
1506
|
+
failures=$((failures + 1))
|
|
1507
|
+
cutover_event_required attempt-failed "$CUTOVER_ROLE" "$current_attempt" "operator control interrupted readiness"
|
|
1508
|
+
if handle_cutover_control; then
|
|
1509
|
+
if [ "$control_result" = "restore" ]; then
|
|
1510
|
+
wd_log "operator control interrupted target readiness — restoring previous"
|
|
1511
|
+
continue
|
|
1512
|
+
fi
|
|
1513
|
+
park_cutover "operator abort follows wait-for-user policy"
|
|
1514
|
+
continue
|
|
1515
|
+
fi
|
|
1516
|
+
fi
|
|
1517
|
+
|
|
383
1518
|
if grep -q 'EADDRINUSE' "$ATTEMPT_LOG" 2>/dev/null; then
|
|
384
1519
|
if grep 'EADDRINUSE' "$ATTEMPT_LOG" | grep -qE "[:.]$PORT([^0-9]|$)"; then
|
|
1520
|
+
if [ -n "$CUTOVER_ID" ]; then
|
|
1521
|
+
# The target never inherits authority to kill an arbitrary owner of
|
|
1522
|
+
# the shared port. Count the attempt so the approved full-spec
|
|
1523
|
+
# recovery policy runs; any listener previously proven inside this
|
|
1524
|
+
# attempt was already retired by kill_current_owned_attempt.
|
|
1525
|
+
wd_log "cutover $CUTOVER_ROLE hit EADDRINUSE on :$PORT — refusing port-based cleanup; counting a $CUTOVER_ROLE failure"
|
|
1526
|
+
readiness_failure_detail="EADDRINUSE on supervised :$PORT; cutover refused port-based cleanup"
|
|
1527
|
+
port_races=5
|
|
1528
|
+
else
|
|
385
1529
|
# The supervised port was still held (a leftover process, a slow exit)
|
|
386
1530
|
# — an operational race, not a code regression. The watchdog owns this
|
|
387
1531
|
# port, so free it and retry WITHOUT counting toward rollback or
|
|
@@ -389,24 +1533,26 @@ while true; do
|
|
|
389
1533
|
# is outside this watchdog's reach and retrying is a hot spin.
|
|
390
1534
|
port_races=$((port_races + 1))
|
|
391
1535
|
if [ "$port_races" -le 5 ]; then
|
|
392
|
-
|
|
1536
|
+
wd_log "boot hit EADDRINUSE on :$PORT — freeing the port and retrying (not a code failure, attempt $port_races/5)"
|
|
393
1537
|
free_port
|
|
394
1538
|
continue
|
|
395
1539
|
fi
|
|
396
|
-
|
|
1540
|
+
wd_log ":$PORT is still held after 5 free attempts — counting this as a boot failure"
|
|
1541
|
+
fi
|
|
397
1542
|
else
|
|
398
1543
|
# EADDRINUSE on a port this watchdog does not own: the start command
|
|
399
1544
|
# targets somewhere else, and freeing :$PORT cannot release it. The
|
|
400
1545
|
# unconditional retry this replaces never counted the attempt, so a
|
|
401
1546
|
# start command aimed at an occupied foreign port respawned the
|
|
402
1547
|
# instance in a tight loop with no backoff and no give-up.
|
|
403
|
-
|
|
1548
|
+
wd_log "boot hit EADDRINUSE on a port other than the supervised :$PORT — the --start command targets a port this watchdog does not own; freeing :$PORT cannot fix that"
|
|
404
1549
|
reset_done=1
|
|
405
1550
|
fi
|
|
406
1551
|
fi
|
|
407
1552
|
|
|
408
1553
|
failures=$((failures + 1))
|
|
409
|
-
|
|
1554
|
+
wd_log "instance failed to come up (failure #$failures: $readiness_failure_detail)"
|
|
1555
|
+
cutover_event_required attempt-failed "$CUTOVER_ROLE" "$current_attempt" "$readiness_failure_detail"
|
|
410
1556
|
|
|
411
1557
|
# The instance came up on a port this watchdog does not own: a start-command
|
|
412
1558
|
# argument, not a code regression. Resetting the checkout cannot change a
|
|
@@ -414,18 +1560,53 @@ while true; do
|
|
|
414
1560
|
# whose subject lives outside the repository) and keep counting toward the
|
|
415
1561
|
# crash page, which is what makes the misconfiguration visible.
|
|
416
1562
|
if [ -n "$bound" ] && ! printf '%s\n' "$bound" | grep -qx "$PORT"; then
|
|
417
|
-
|
|
1563
|
+
wd_log "instance bound :$(printf '%s' "$bound" | paste -sd, -) but supervision owns :$PORT — the --start command does not bind the supervised port; a repository rollback cannot fix that"
|
|
418
1564
|
reset_done=1
|
|
419
1565
|
fi
|
|
420
1566
|
|
|
421
|
-
|
|
1567
|
+
# A launch cutover recovers the complete previous spec or waits, exactly as
|
|
1568
|
+
# approved before the stop. It never falls through to the ordinary
|
|
1569
|
+
# repository/composition reset machinery: neither can repair a command,
|
|
1570
|
+
# home, credential repo, host root, or profile change as one unit.
|
|
1571
|
+
if [ -n "$CUTOVER_ID" ] && [ "$failures" -ge "$TARGET_FAILURE_LIMIT" ]; then
|
|
1572
|
+
if [ "$CUTOVER_ROLE" = "target" ] && [ "$CUTOVER_POLICY" = "restore-previous" ] \
|
|
1573
|
+
&& [ -n "$PREVIOUS_START" ] && [ -n "$PREVIOUS_HOME" ] && [ -n "$PREVIOUS_REPO" ] \
|
|
1574
|
+
&& [ -n "$PREVIOUS_HARNESS_ROOT" ]; then
|
|
1575
|
+
wd_log "target launch failed after $failures attempt(s) — restoring the approved previous launch specification"
|
|
1576
|
+
if select_previous_spec "target failed after $failures attempt(s); restoring previous spec"; then
|
|
1577
|
+
continue
|
|
1578
|
+
fi
|
|
1579
|
+
cutover_event_required awaiting-user "target failed and filesystem transition rollback did not complete; previous was not started"
|
|
1580
|
+
park_cutover "target failed; previous restore is blocked by filesystem transition rollback"
|
|
1581
|
+
continue
|
|
1582
|
+
fi
|
|
1583
|
+
wd_log "launch cutover cannot become ready — approved policy is ${CUTOVER_POLICY:-wait-for-user}; parking for user action"
|
|
1584
|
+
cutover_event_required awaiting-user "$CUTOVER_ROLE launch failed after $failures attempt(s)"
|
|
1585
|
+
printf '%s launch cutover waiting after %s failures\n' "$(date '+%F %T')" "$failures" > "$GIVE_UP_MARKER"
|
|
1586
|
+
WD_PORT="$PORT" WD_PID="$$" node -e "$(page_script)" &
|
|
1587
|
+
page_pid=$!
|
|
1588
|
+
wait "$page_pid"
|
|
1589
|
+
page_pid=''
|
|
1590
|
+
continue
|
|
1591
|
+
fi
|
|
1592
|
+
|
|
1593
|
+
if [ -z "$CUTOVER_ID" ] && [ "$failures" -ge 2 ] && [ "$reset_done" -eq 0 ]; then
|
|
422
1594
|
sha=$(rollback_sha)
|
|
423
1595
|
if [ -z "$sha" ]; then
|
|
424
|
-
|
|
1596
|
+
wd_log "no guard credential/checkpoint recorded; cannot roll back"
|
|
425
1597
|
elif failure_subject_outside_repo; then
|
|
426
1598
|
# The failure lives outside the checkout (profile overlay, installed
|
|
427
|
-
# plugin, environment) — reverting the repository cannot fix it.
|
|
428
|
-
|
|
1599
|
+
# plugin, environment) — reverting the repository cannot fix it. But
|
|
1600
|
+
# the COMPOSITION can be rolled back: if a healthy-boot snapshot of
|
|
1601
|
+
# the profile inputs exists and differs from the live one, restore it
|
|
1602
|
+
# (unmounting the newest plugin change) and retry with a clean count.
|
|
1603
|
+
if [ "$comp_restore_done" -eq 0 ] && restore_composition; then
|
|
1604
|
+
comp_restore_done=1
|
|
1605
|
+
comp_restored=1
|
|
1606
|
+
failures=0
|
|
1607
|
+
continue
|
|
1608
|
+
fi
|
|
1609
|
+
wd_log "boot failure originates outside $REPO — repository rollback cannot fix it; leaving the checkout untouched"
|
|
429
1610
|
reset_done=1
|
|
430
1611
|
else
|
|
431
1612
|
if rollback_to "$sha"; then
|
|
@@ -440,36 +1621,157 @@ while true; do
|
|
|
440
1621
|
fi
|
|
441
1622
|
|
|
442
1623
|
if [ "$failures" -ge 4 ]; then
|
|
443
|
-
|
|
1624
|
+
wd_log "giving up after $failures consecutive failures"
|
|
444
1625
|
printf '%s giving up after %s failures\n' "$(date '+%F %T')" "$failures" > "$GIVE_UP_MARKER"
|
|
445
|
-
|
|
1626
|
+
wd_log "serving crash page on :$PORT — click 重试 or send SIGUSR1 to $$"
|
|
446
1627
|
WD_PORT="$PORT" WD_PID="$$" node -e "$(page_script)" &
|
|
447
1628
|
page_pid=$!
|
|
448
1629
|
wait "$page_pid"
|
|
1630
|
+
page_pid=''
|
|
449
1631
|
continue
|
|
450
1632
|
fi
|
|
451
1633
|
|
|
452
|
-
|
|
1634
|
+
wd_sleep $((failures * 5))
|
|
453
1635
|
continue
|
|
454
1636
|
fi
|
|
455
1637
|
|
|
456
1638
|
# Instance is up.
|
|
457
|
-
|
|
458
|
-
|
|
1639
|
+
wd_log "instance ready on :$PORT ($readiness_detail) — instance output: $ATTEMPT_LOG"
|
|
1640
|
+
if [ "$comp_restored" = "1" ]; then
|
|
1641
|
+
# The boot only succeeded because the composition was rolled back — the
|
|
1642
|
+
# recovery (newest plugin change unmounted) must be reported, not silent.
|
|
1643
|
+
guard_cmd record-composition-recovery --state-dir "$STATE_DIR" --detail "$comp_restore_detail"
|
|
1644
|
+
comp_restored=0
|
|
1645
|
+
fi
|
|
459
1646
|
|
|
460
1647
|
# Intentional restart: run the guard canary (credential fresh + HEAD match).
|
|
461
1648
|
if [ -f "$RESTART_MARKER" ]; then
|
|
462
|
-
|
|
463
|
-
|
|
1649
|
+
canary_proven=0
|
|
1650
|
+
if guard_verify && current_ownership_matches; then
|
|
1651
|
+
cutover_event_required canary "$CUTOVER_ROLE" pass
|
|
1652
|
+
# Persisting canary evidence can take long enough for a short-lived child
|
|
1653
|
+
# to exit. Recheck after the event and before the terminal ready event.
|
|
1654
|
+
if current_ownership_matches; then canary_proven=1; fi
|
|
1655
|
+
fi
|
|
1656
|
+
if [ "$canary_proven" = "1" ]; then
|
|
1657
|
+
if [ -n "$CUTOVER_ID" ] && ! complete_browser_handoff; then
|
|
1658
|
+
failures=$((failures + 1))
|
|
1659
|
+
cutover_event_required attempt-failed "$CUTOVER_ROLE" "$current_attempt" \
|
|
1660
|
+
"browser handoff failed after canary and stable ownership"
|
|
1661
|
+
if [ "$CUTOVER_ROLE" = "target" ] && [ "$CUTOVER_POLICY" = "restore-previous" ] \
|
|
1662
|
+
&& [ -n "$PREVIOUS_START" ] && [ -n "$PREVIOUS_HOME" ] && [ -n "$PREVIOUS_REPO" ] \
|
|
1663
|
+
&& [ -n "$PREVIOUS_HARNESS_ROOT" ]; then
|
|
1664
|
+
kill_current_owned_attempt
|
|
1665
|
+
wait "$child" 2>/dev/null || true
|
|
1666
|
+
if select_previous_spec "target canary passed but browser handoff failed; restoring previous spec"; then
|
|
1667
|
+
continue
|
|
1668
|
+
fi
|
|
1669
|
+
cutover_event_required awaiting-user "browser handoff failed and filesystem transition rollback did not complete; previous was not started"
|
|
1670
|
+
park_cutover "browser handoff failed; previous restore is blocked by filesystem transition rollback"
|
|
1671
|
+
continue
|
|
1672
|
+
fi
|
|
1673
|
+
if [ "$CUTOVER_ROLE" = "previous" ]; then
|
|
1674
|
+
rm -f "$RESTART_MARKER"
|
|
1675
|
+
cutover_event_required awaiting-user "restored previous host is ready, but browser handoff was not acknowledged"
|
|
1676
|
+
wait_cutover_with_live_child
|
|
1677
|
+
continue
|
|
1678
|
+
fi
|
|
1679
|
+
kill_current_owned_attempt
|
|
1680
|
+
wait "$child" 2>/dev/null || true
|
|
1681
|
+
rm -f "$RESTART_MARKER"
|
|
1682
|
+
cutover_event_required awaiting-user "target browser handoff failed after canary"
|
|
1683
|
+
printf '%s launch cutover waiting after browser handoff failure\n' "$(date '+%F %T')" > "$GIVE_UP_MARKER"
|
|
1684
|
+
WD_PORT="$PORT" WD_PID="$$" node -e "$(page_script)" &
|
|
1685
|
+
page_pid=$!
|
|
1686
|
+
wait "$page_pid"
|
|
1687
|
+
page_pid=''
|
|
1688
|
+
continue
|
|
1689
|
+
fi
|
|
1690
|
+
wd_log "canary PASS — browser handoff settled; recording deployment proof"
|
|
1691
|
+
if [ -n "$CUTOVER_ID" ]; then
|
|
1692
|
+
cutover_event_required ready "$CUTOVER_ROLE"
|
|
1693
|
+
CUTOVER_ID=''
|
|
1694
|
+
fi
|
|
1695
|
+
if ! guard_cmd record-proven-deployment --repo "$REPO" --state-dir "$STATE_DIR"; then
|
|
1696
|
+
# Proof persistence is an optimization for a later pure restart. This
|
|
1697
|
+
# boot already passed readiness, ownership, and canary; keep it up but
|
|
1698
|
+
# force the next restart back through fresh build/test evidence.
|
|
1699
|
+
wd_log "deployment proof unavailable — the next restart requires fresh build/test evidence" >&2
|
|
1700
|
+
fi
|
|
464
1701
|
rm -f "$RESTART_MARKER"
|
|
465
1702
|
else
|
|
466
|
-
|
|
1703
|
+
if [ -n "$CUTOVER_ID" ]; then
|
|
1704
|
+
wd_log "canary/ownership FAIL during launch cutover"
|
|
1705
|
+
if [ "$CUTOVER_ROLE" = "target" ]; then
|
|
1706
|
+
cutover_event_required canary target fail "credential/head verification or child/listener identity failed after readiness"
|
|
1707
|
+
elif current_ownership_matches; then
|
|
1708
|
+
# The singleton restart credential normally belongs to the rejected
|
|
1709
|
+
# target repo. Do not relabel that target credential failure as a
|
|
1710
|
+
# previous-host canary failure: stable process/listener ownership is
|
|
1711
|
+
# the explicit recovery proof when no previous-scoped credential is
|
|
1712
|
+
# available.
|
|
1713
|
+
cutover_event_required canary previous skipped \
|
|
1714
|
+
"target-scoped credential is not previous recovery evidence; stable ownership remained proven"
|
|
1715
|
+
else
|
|
1716
|
+
cutover_event_required canary previous fail \
|
|
1717
|
+
"restored child/listener identity failed after readiness"
|
|
1718
|
+
fi
|
|
1719
|
+
if [ "$CUTOVER_ROLE" = "target" ] && [ "$CUTOVER_POLICY" = "restore-previous" ] \
|
|
1720
|
+
&& [ -n "$PREVIOUS_START" ] && [ -n "$PREVIOUS_HOME" ] && [ -n "$PREVIOUS_REPO" ] \
|
|
1721
|
+
&& [ -n "$PREVIOUS_HARNESS_ROOT" ]; then
|
|
1722
|
+
kill_current_owned_attempt
|
|
1723
|
+
wait "$child" 2>/dev/null || true
|
|
1724
|
+
if select_previous_spec "target became ready but canary failed; restoring previous spec"; then
|
|
1725
|
+
continue
|
|
1726
|
+
fi
|
|
1727
|
+
cutover_event_required awaiting-user "target canary failed and filesystem transition rollback did not complete; previous was not started"
|
|
1728
|
+
park_cutover "target canary failed; previous restore is blocked by filesystem transition rollback"
|
|
1729
|
+
continue
|
|
1730
|
+
fi
|
|
1731
|
+
if [ "$CUTOVER_ROLE" = "previous" ]; then
|
|
1732
|
+
if ! current_ownership_matches; then
|
|
1733
|
+
failures=$((failures + 1))
|
|
1734
|
+
cutover_event_required attempt-failed previous "$current_attempt" \
|
|
1735
|
+
"restored previous ownership changed after readiness"
|
|
1736
|
+
cutover_event_required awaiting-user "restored previous host lost child/listener ownership"
|
|
1737
|
+
wait_cutover_with_live_child
|
|
1738
|
+
continue
|
|
1739
|
+
fi
|
|
1740
|
+
# The previous service is restored and ready; a credential tied to a
|
|
1741
|
+
# different target repo may legitimately fail. Browser acknowledgement
|
|
1742
|
+
# is still required before this recovery becomes terminal.
|
|
1743
|
+
rm -f "$RESTART_MARKER"
|
|
1744
|
+
if complete_browser_handoff; then
|
|
1745
|
+
cutover_event_required ready previous
|
|
1746
|
+
CUTOVER_ID=''
|
|
1747
|
+
else
|
|
1748
|
+
failures=$((failures + 1))
|
|
1749
|
+
cutover_event_required attempt-failed previous "$current_attempt" \
|
|
1750
|
+
"browser handoff failed after restored-previous canary settled"
|
|
1751
|
+
cutover_event_required awaiting-user "restored previous host is ready, but browser handoff was not acknowledged"
|
|
1752
|
+
wait_cutover_with_live_child
|
|
1753
|
+
continue
|
|
1754
|
+
fi
|
|
1755
|
+
else
|
|
1756
|
+
kill_current_owned_attempt
|
|
1757
|
+
wait "$child" 2>/dev/null || true
|
|
1758
|
+
cutover_event_required awaiting-user "target canary failed"
|
|
1759
|
+
printf '%s launch cutover waiting after canary failure\n' "$(date '+%F %T')" > "$GIVE_UP_MARKER"
|
|
1760
|
+
WD_PORT="$PORT" WD_PID="$$" node -e "$(page_script)" &
|
|
1761
|
+
page_pid=$!
|
|
1762
|
+
wait "$page_pid"
|
|
1763
|
+
page_pid=''
|
|
1764
|
+
continue
|
|
1765
|
+
fi
|
|
1766
|
+
else
|
|
1767
|
+
wd_log "canary FAIL — rolling back to last known-good"
|
|
467
1768
|
sha=$(rollback_sha)
|
|
468
1769
|
if [ -n "$sha" ]; then rollback_to "$sha" || true; fi
|
|
469
1770
|
rm -f "$RESTART_MARKER"
|
|
470
1771
|
failures=0
|
|
471
1772
|
reset_done=0
|
|
472
1773
|
continue
|
|
1774
|
+
fi
|
|
473
1775
|
fi
|
|
474
1776
|
else
|
|
475
1777
|
# Unplanned exit (crash, or a stop outside the guard): leave a record the
|
|
@@ -489,6 +1791,12 @@ while true; do
|
|
|
489
1791
|
fi
|
|
490
1792
|
fi
|
|
491
1793
|
|
|
1794
|
+
# Only a fully ready + canary-settled boot becomes the deployment rollback
|
|
1795
|
+
# target. Stamping before the cutover canary once made a rejected target the
|
|
1796
|
+
# very revision ordinary rollback preferred.
|
|
1797
|
+
stamp_last_good_boot
|
|
1798
|
+
snapshot_composition
|
|
1799
|
+
|
|
492
1800
|
failures=0
|
|
493
1801
|
reset_done=0
|
|
494
1802
|
port_races=0
|
|
@@ -501,24 +1809,24 @@ while true; do
|
|
|
501
1809
|
if [ ! -f "$PIDFILE" ]; then (set -C; echo $$ > "$PIDFILE") 2>/dev/null || true; fi
|
|
502
1810
|
pidowner=$(cat "$PIDFILE" 2>/dev/null)
|
|
503
1811
|
if [ -n "$pidowner" ] && [ "$pidowner" != "$$" ] && kill -0 "$pidowner" 2>/dev/null; then
|
|
504
|
-
|
|
1812
|
+
wd_log "pidfile now owned by live pid $pidowner — yielding (instance left running for the new owner)"
|
|
505
1813
|
yielded=1
|
|
506
|
-
exit
|
|
1814
|
+
exit 75
|
|
507
1815
|
fi
|
|
508
1816
|
fi
|
|
509
|
-
|
|
1817
|
+
wd_sleep 2
|
|
510
1818
|
done
|
|
511
1819
|
wait "$child"
|
|
512
1820
|
|
|
513
1821
|
# Explicit stop: exit the watchdog without respawn.
|
|
514
1822
|
if [ -f "$STOP_MARKER" ]; then
|
|
515
|
-
|
|
1823
|
+
wd_log "stop marker present — exiting"
|
|
516
1824
|
rm -f "$STOP_MARKER" "$PIDFILE"
|
|
517
1825
|
exit 0
|
|
518
1826
|
fi
|
|
519
1827
|
|
|
520
|
-
|
|
521
|
-
if
|
|
1828
|
+
wd_sleep 3
|
|
1829
|
+
if transport_up; then
|
|
522
1830
|
free_port
|
|
523
1831
|
fi
|
|
524
1832
|
done
|