@khorsheed/dsh-ankh-guard 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/README.en.md +75 -29
  3. package/README.i18n.yaml +2 -2
  4. package/README.md +74 -29
  5. package/lib/cli.js +2383 -209
  6. package/lib/client.js +257 -0
  7. package/lib/exit-agent.js +5 -2
  8. package/lib/index.js +722 -38
  9. package/lib/invariant.js +1 -1
  10. package/lib/preflight-runner.js +125 -47
  11. package/lib/processes-BjZgJjQr.js +344 -0
  12. package/lib/restart-context-D6nISh28.js +1245 -0
  13. package/lib/restart-context-DUyExi9O.js +1245 -0
  14. package/lib/{state-Dhx9VG44.js → state-4f7yny39.js} +60 -13
  15. package/lib/state-CZMypGkB.js +323 -0
  16. package/lib/test-seam-DnvLWTeO.js +119 -0
  17. package/lib/test-seam-cli.js +24 -0
  18. package/lib/test-seam-dwvaKjRp.js +459 -0
  19. package/lib/test-seam.js +2 -0
  20. package/lib/types/browser-handoff.d.ts +55 -0
  21. package/lib/types/browser-handoff.js +489 -0
  22. package/lib/types/cli.d.ts +34 -4
  23. package/lib/types/cli.js +1487 -225
  24. package/lib/types/client/index.d.ts +15 -0
  25. package/lib/types/client/index.js +264 -0
  26. package/lib/types/deployment-proof.d.ts +24 -0
  27. package/lib/types/deployment-proof.js +314 -0
  28. package/lib/types/exit-agent.js +2 -0
  29. package/lib/types/git.d.ts +12 -3
  30. package/lib/types/git.js +69 -7
  31. package/lib/types/index.d.ts +66 -3
  32. package/lib/types/index.js +157 -39
  33. package/lib/types/launch-spec.d.ts +263 -0
  34. package/lib/types/launch-spec.js +823 -0
  35. package/lib/types/preflight-runner.d.ts +23 -12
  36. package/lib/types/preflight-runner.js +152 -57
  37. package/lib/types/processes.d.ts +38 -6
  38. package/lib/types/processes.js +236 -10
  39. package/lib/types/restart-context.d.ts +50 -0
  40. package/lib/types/restart-context.js +106 -0
  41. package/lib/types/restart-request.d.ts +32 -0
  42. package/lib/types/restart-request.js +128 -0
  43. package/lib/types/state-files.d.ts +30 -0
  44. package/lib/types/state-files.js +55 -0
  45. package/lib/types/state.d.ts +29 -2
  46. package/lib/types/state.js +52 -7
  47. package/lib/types/temp-artifact.d.ts +15 -0
  48. package/lib/types/temp-artifact.js +17 -0
  49. package/lib/types/test-seam-cli.d.ts +3 -0
  50. package/lib/types/test-seam-cli.js +27 -0
  51. package/lib/types/test-seam.d.ts +55 -0
  52. package/lib/types/test-seam.js +112 -0
  53. package/lib/types/transition.d.ts +118 -0
  54. package/lib/types/transition.js +717 -0
  55. package/package.json +29 -9
  56. package/scripts/dsh-watchdog.sh +1388 -80
  57. package/scripts/install-launchd.sh +43 -5
  58. package/scripts/install-systemd.sh +43 -5
  59. package/scripts/on-install.js +1 -1
  60. package/skills/dsh-self-restart-guard/SKILL.md +38 -12
  61. package/lib/processes-hCAmwma-.js +0 -127
  62. package/lib/restart-context-DmnQXNf-.js +0 -421
@@ -28,8 +28,12 @@
28
28
  # WD_HOME=DIR dsh root (default: $DSH_HOME)
29
29
  # WD_STATE_DIR=DIR state dir: markers, pidfile, logs (default: <WD_HOME>/state)
30
30
  # WD_PORT=N port to own (default 3080)
31
- # WD_REPO=DIR checkout the guard rollback operates on
31
+ # WD_REPO=DIR credential/rollback repository (not the host checkout)
32
+ # WD_HARNESS_ROOT=DIR host checkout exported to the child as DSH_HARNESS
32
33
  # WD_START="CMD" shell command that starts the supervised instance
34
+ # WD_PROFILE=NAME the profile the instance boots (default web) — its
35
+ # composition inputs are snapshotted at healthy boots and
36
+ # restored when boot failures originate outside the repo
33
37
  # WD_GUARD="CMD" how to invoke the guard CLI (default: dsh-ankh-guard)
34
38
  # WD_WAIT_OWNER=1 don't adopt the port; wait for the current owner to exit
35
39
  # WD_DELAY=N sleep N seconds before adopting/observing the port
@@ -38,7 +42,21 @@
38
42
  # WD_ADOPTION=1 the CLI saw a live owner at supervise time — the first
39
43
  # boot is a takeover (report it), not a first-ever boot
40
44
  # WD_SUPERVISE=1 write/check the pidfile (one watchdog only)
41
- # WD_BOOT_TIMEOUT=N seconds to wait for the port to answer 200 (default 60)
45
+ # WD_BOOT_TIMEOUT=N seconds to prove application readiness (default 60)
46
+ # WD_TAKEOVER_FROM=P replace this live watchdog's pidfile claim before the
47
+ # old instance is interrupted (launch cutover only)
48
+ # WD_CUTOVER_ID=ID durable launch-cutover receipt transaction
49
+ # WD_CUTOVER_POLICY= restore-previous or wait-for-user (approved pre-stop)
50
+ # WD_CUTOVER_DELAY_SECONDS=N grace after supervisor claim before old-child stop
51
+ # WD_PREVIOUS_CHILD_*=PID/start token authoritative old supervisor child root
52
+ # WD_PREVIOUS_LISTENER_*=PID/start token listener inside that old child tree
53
+ # WD_READY_STABILITY_SECONDS=N unchanged child/listener proof window (default 3)
54
+ # WD_BROWSER_HANDOFF=required|off after an authenticated launch-URL exchange
55
+ # WD_TRANSITION_PLAN_SHA256=SHA-256 signals a prepared filesystem transition
56
+ # bound to the active cutover; the guard CLI reads the
57
+ # durable plan and journal rather than trusting this value
58
+ # WD_BROWSER_HANDOFF_TIMEOUT_SECONDS=N wait for original-tab acknowledgement,
59
+ # then (after fallback open) for fallback acknowledgement
42
60
  # WD_TEST_FAKE=1 launch a throwaway http server instead of the instance
43
61
  # WD_TEST_BREAK=1 launch a command that always fails (give-up testing)
44
62
  #
@@ -55,6 +73,26 @@ PORT="${WD_PORT:-3080}"
55
73
  DELAY="${WD_DELAY:-0}"
56
74
  BOOT_TIMEOUT="${WD_BOOT_TIMEOUT:-60}"
57
75
  REPO="${WD_REPO:-}"
76
+ HARNESS_ROOT="${WD_HARNESS_ROOT:-${DSH_HARNESS:-}}"
77
+ PROFILE="${WD_PROFILE:-web}"
78
+ START_CMD="${WD_START:-}"
79
+ CUTOVER_ID="${WD_CUTOVER_ID:-}"
80
+ CUTOVER_POLICY="${WD_CUTOVER_POLICY:-}"
81
+ CUTOVER_ROLE="${WD_CUTOVER_ROLE:-target}"
82
+ PREVIOUS_START="${WD_PREVIOUS_START:-}"
83
+ PREVIOUS_HOME="${WD_PREVIOUS_HOME:-}"
84
+ PREVIOUS_REPO="${WD_PREVIOUS_REPO:-}"
85
+ PREVIOUS_HARNESS_ROOT="${WD_PREVIOUS_HARNESS_ROOT:-}"
86
+ PREVIOUS_PROFILE="${WD_PREVIOUS_PROFILE:-}"
87
+ PREVIOUS_CHILD_PID="${WD_PREVIOUS_CHILD_PID:-}"
88
+ PREVIOUS_CHILD_START="${WD_PREVIOUS_CHILD_START:-}"
89
+ PREVIOUS_LISTENER_PID="${WD_PREVIOUS_LISTENER_PID:-}"
90
+ PREVIOUS_LISTENER_START="${WD_PREVIOUS_LISTENER_START:-}"
91
+ BROWSER_HANDOFF="${WD_BROWSER_HANDOFF:-off}"
92
+ TARGET_FAILURE_LIMIT="${WD_TARGET_FAILURE_LIMIT:-2}"
93
+ READY_STABILITY_SECONDS="${WD_READY_STABILITY_SECONDS:-3}"
94
+ BROWSER_HANDOFF_TIMEOUT_SECONDS="${WD_BROWSER_HANDOFF_TIMEOUT_SECONDS:-8}"
95
+ TRANSITION_PLAN_SHA256="${WD_TRANSITION_PLAN_SHA256:-}"
58
96
  # Every marker, the pidfile, and the attempt log live in ONE state directory:
59
97
  # WD_STATE_DIR when the guard CLI names it (its --state-dir), else the
60
98
  # conventional <home>/state. Deriving it here as <home>/state while the guard
@@ -67,44 +105,746 @@ RESTART_MARKER="$STATE_DIR/restart-requested.json"
67
105
  STOP_MARKER="$STATE_DIR/watchdog-stop"
68
106
  PIDFILE="$STATE_DIR/watchdog.pid"
69
107
  ATTEMPT_LOG="$STATE_DIR/boot-attempt.log"
108
+ CONTROL_FILE="$STATE_DIR/launch-cutover-control.json"
109
+ CONTROL_ABORT_FILE="$STATE_DIR/launch-cutover-abort.json"
110
+ CONTROL_RESTORE_FILE="$STATE_DIR/launch-cutover-restore-previous.json"
111
+ BROWSER_HANDOFF_REQUEST_FILE="$STATE_DIR/browser-handoff-request.json"
112
+ BROWSER_HANDOFF_ACK_FILE="$STATE_DIR/browser-handoff-ack.json"
70
113
 
71
- [ -n "$DSH_ROOT" ] || { echo "[watchdog] WD_HOME or DSH_HOME must be set" >&2; exit 1; }
114
+ # Timestamp every lifecycle line so a durable receipt can be correlated with
115
+ # supervisor/child PIDs across launchd, systemd, and detached CLI restarts.
116
+ wd_log() {
117
+ printf '%s [watchdog] %s\n' "$(date '+%Y-%m-%dT%H:%M:%S%z')" "$*"
118
+ }
119
+
120
+ resolve_tool() {
121
+ local name=$1 candidate
122
+ shift
123
+ for candidate in "$@"; do [ -x "$candidate" ] && { printf '%s' "$candidate"; return 0; }; done
124
+ command -v "$name" 2>/dev/null || return 1
125
+ }
126
+
127
+ # macOS tool sessions commonly omit /usr/sbin from PATH. Listener identity is
128
+ # a safety proof, not optional telemetry, so resolve canonical absolute paths
129
+ # and fail loud when the proof machinery truly is unavailable.
130
+ LSOF_BIN=$(resolve_tool lsof /usr/sbin/lsof /usr/bin/lsof) || { wd_log "lsof is required to prove listener ownership" >&2; exit 1; }
131
+ PS_BIN=$(resolve_tool ps /bin/ps /usr/bin/ps) || { wd_log "ps is required to prove process identity" >&2; exit 1; }
132
+ PGREP_BIN=$(resolve_tool pgrep /usr/bin/pgrep /bin/pgrep) || { wd_log "pgrep is required to manage the supervised child tree" >&2; exit 1; }
133
+ SYSCTL_BIN=$(resolve_tool sysctl /usr/sbin/sysctl /sbin/sysctl 2>/dev/null || true)
134
+ PYTHON_BIN=$(resolve_tool python3 /usr/bin/python3 /opt/homebrew/bin/python3 2>/dev/null || true)
135
+
136
+ [ -n "$DSH_ROOT" ] || { wd_log "WD_HOME or DSH_HOME must be set" >&2; exit 1; }
72
137
  export DSH_HOME="$DSH_ROOT"
73
138
  mkdir -p "$STATE_DIR"
74
- cd "$DSH_ROOT/home" 2>/dev/null || cd /tmp || exit 1
139
+
140
+ # Production sleeps retain their exact durations. A test run may scale only
141
+ # internal polling/backoff after presenting the private run coordinates used
142
+ # by the ownership ledger; the scale is never persisted in a launch spec.
143
+ wd_sleep() {
144
+ local duration=$1 scale=${ANKH_GUARD_TEST_SLEEP_SCALE:-1}
145
+ if [ -n "${ANKH_GUARD_TEST_RUN_DIR:-}" ] && [ -n "${ANKH_GUARD_TEST_RUN_TOKEN:-}" ] \
146
+ && printf '%s' "$scale" | grep -Eq '^0\.[0-9]+$|^1(\.0+)?$'; then
147
+ duration=$(/usr/bin/awk -v duration="$duration" -v scale="$scale" \
148
+ 'BEGIN { value=duration*scale; if (value < 0.01) value=0.01; printf "%.3f", value }')
149
+ fi
150
+ sleep "$duration"
151
+ }
75
152
 
76
153
  launch_instance() {
77
- if [ "${WD_TEST_BREAK:-0}" = "1" ]; then sleep 1; exit 1; fi
154
+ test_register_self instance-wrapper
155
+ test_event_self instance-wrapper process-started
156
+ if [ "${WD_TEST_BREAK:-0}" = "1" ]; then wd_sleep 1; exit 1; fi
78
157
  if [ "${WD_TEST_FAKE:-0}" = "1" ]; then
79
158
  node -e "require('http').createServer((q,s)=>s.end('ok')).listen($PORT,'127.0.0.1')"
80
159
  exit
81
160
  fi
82
- if [ -z "${WD_START:-}" ]; then echo "[watchdog] WD_START unset — nothing to supervise" >&2; exit 1; fi
83
- sh -c "$WD_START"
161
+ if [ -z "$START_CMD" ]; then wd_log "launch command unset — nothing to supervise" >&2; exit 1; fi
162
+ # The instance inherits this process's environment: scrub EVERY WD_* so no
163
+ # supervision variable can leak into the shells the instance hosts. A leaked
164
+ # WD_STATE_DIR retargets any watchdog script those shells spawn (observed
165
+ # 2026-08-29: an agent session inside the supervised deployment ran the test
166
+ # suite straight into the PROD state dir — racers yielded to the live
167
+ # pidfile owner, and the reclaim case never saw its temp pidfile). The scrub
168
+ # is prefix-based, not a name list: the CLI adds WD_* variables over time
169
+ # (WD_GUARD, WD_WAIT_OWNER, WD_ADOPTION, …) and a list silently goes stale.
170
+ # WD_START is captured first: the unset would otherwise eat the command
171
+ # itself. The guard CLI's bare-restart spawn applies the same scrub.
172
+ local start_cmd=$START_CMD launch_home=$DSH_ROOT launch_harness_root=$HARNESS_ROOT
173
+ (
174
+ export DSH_HOME="$launch_home"
175
+ if [ -n "$launch_harness_root" ]; then export DSH_HARNESS="$launch_harness_root"; fi
176
+ cd "$launch_home/home" 2>/dev/null || cd /tmp || exit 1
177
+ for v in $(env | sed -n 's/^\(WD_[^=]*\)=.*/\1/p'); do unset "$v"; done
178
+ sh -c "$start_cmd"
179
+ )
180
+ }
181
+
182
+ http_status() {
183
+ curl -s --noproxy '*' -o /dev/null -w '%{http_code}' --max-time 3 "http://127.0.0.1:$PORT/" 2>/dev/null || true
184
+ }
185
+
186
+ cutover_event() {
187
+ [ -n "$CUTOVER_ID" ] || return 0
188
+ guard_cmd cutover-event "$CUTOVER_ID" "$@" --state-dir "$STATE_DIR" >/dev/null 2>&1
189
+ }
190
+
191
+ # Receipt updates are part of the transaction, not telemetry. Keep the proven
192
+ # child (or the still-running old child during supervisor handoff) available
193
+ # while retrying a transient state/CLI failure; never advance in memory past a
194
+ # durable event that crash recovery depends on.
195
+ cutover_event_required() {
196
+ [ -n "$CUTOVER_ID" ] || return 0
197
+ while ! cutover_event "$@"; do
198
+ wd_log "could not persist cutover event $1 — retrying; service state is unchanged" >&2
199
+ wd_sleep 1
200
+ done
201
+ }
202
+
203
+ cutover_control_action() {
204
+ [ -n "$CUTOVER_ID" ] || return 0
205
+ node -e '
206
+ const fs = require("fs")
207
+ const [restoreFile, abortFile, legacyFile, id] = process.argv.slice(1)
208
+ const read = (file) => {
209
+ try {
210
+ const value = JSON.parse(fs.readFileSync(file, "utf8"))
211
+ return value?.version === 1 && value.cutoverId === id
212
+ && (value.action === "abort" || value.action === "restore-previous") ? value.action : ""
213
+ } catch { return "" }
214
+ }
215
+ const restore = read(restoreFile)
216
+ const legacy = read(legacyFile)
217
+ process.stdout.write(restore === "restore-previous" || legacy === "restore-previous"
218
+ ? "restore-previous" : read(abortFile) || legacy)
219
+ ' "$CONTROL_RESTORE_FILE" "$CONTROL_ABORT_FILE" "$CONTROL_FILE" "$CUTOVER_ID" 2>/dev/null
220
+ }
221
+
222
+ # The host owns the shape of its per-process launch URL. Discovery is generic:
223
+ # the first HTTP URL printed by THIS attempt whose authority is exactly the
224
+ # supervised loopback authority and whose query is non-empty. No parameter
225
+ # name ("token" or otherwise) is part of the watchdog protocol.
226
+ launch_url_from_output() {
227
+ node -e '
228
+ const fs = require("fs")
229
+ const [file, port] = process.argv.slice(1)
230
+ let text = ""
231
+ try { text = fs.readFileSync(file, "utf8") } catch {}
232
+ for (const match of text.matchAll(/https?:\/\/[^\s)]+/g)) {
233
+ try {
234
+ const url = new URL(match[0])
235
+ if (url.protocol === "http:" && url.hostname === "127.0.0.1"
236
+ && url.port === port && url.pathname === "/" && url.search !== ""
237
+ && url.username === "" && url.password === "" && url.hash === "") {
238
+ process.stdout.write(url.href)
239
+ break
240
+ }
241
+ } catch {}
242
+ }
243
+ ' "$ATTEMPT_LOG" "$PORT" 2>/dev/null
244
+ }
245
+
246
+ # The process output is durable operational evidence, but a launch URL is a
247
+ # bearer credential. Once captured in memory, overwrite every matching URL in
248
+ # place with an equal-length marker. Equal length preserves the active child's
249
+ # append offset; an atomic rename here would strand later output on an unlinked
250
+ # inode. The failure path calls this too, so a process that prints then exits
251
+ # cannot have its credential mirrored into the watchdog log.
252
+ redact_launch_urls_in_output() {
253
+ node -e '
254
+ const fs = require("fs")
255
+ const [file, port] = process.argv.slice(1)
256
+ let text
257
+ try { text = fs.readFileSync(file, "utf8") } catch { process.exit(0) }
258
+ const edits = []
259
+ for (const match of text.matchAll(/https?:\/\/[^\s)]+/g)) {
260
+ try {
261
+ const url = new URL(match[0])
262
+ if (url.protocol === "http:" && url.hostname === "127.0.0.1"
263
+ && url.port === port && url.pathname === "/" && url.search !== ""
264
+ && url.username === "" && url.password === "" && url.hash === "") {
265
+ const start = Buffer.byteLength(text.slice(0, match.index))
266
+ const length = Buffer.byteLength(match[0])
267
+ edits.push({ start, length })
268
+ }
269
+ } catch {}
270
+ }
271
+ if (edits.length === 0) process.exit(0)
272
+ const fd = fs.openSync(file, "r+")
273
+ try {
274
+ for (const { start, length } of edits) {
275
+ const label = Buffer.from("[launch-url-redacted]")
276
+ const replacement = Buffer.alloc(length, 0x20)
277
+ label.copy(replacement, 0, 0, Math.min(label.length, replacement.length))
278
+ fs.writeSync(fd, replacement, 0, replacement.length, start)
279
+ }
280
+ } finally { fs.closeSync(fd) }
281
+ ' "$ATTEMPT_LOG" "$PORT" 2>/dev/null || true
282
+ }
283
+
284
+ open_launch_url() {
285
+ local url=$1
286
+ if [ -n "${WD_BROWSER_OPEN_COMMAND:-}" ]; then
287
+ "${WD_BROWSER_OPEN_COMMAND}" "$url" >/dev/null 2>&1
288
+ elif command -v open >/dev/null 2>&1; then
289
+ open "$url" >/dev/null 2>&1
290
+ elif command -v xdg-open >/dev/null 2>&1; then
291
+ xdg-open "$url" >/dev/null 2>&1
292
+ elif command -v gio >/dev/null 2>&1; then
293
+ gio open "$url" >/dev/null 2>&1
294
+ else
295
+ return 1
296
+ fi
297
+ }
298
+
299
+ last_transport_status=''
300
+ launch_url_reported=0
301
+ launch_url_value=''
302
+ browser_handoff_done=0
303
+ browser_handoff_reported=0
304
+ handoff_cookie_jar=''
305
+ readiness_detail=''
306
+ protected_ready=0
307
+
308
+ # Readiness has two layers. Any HTTP response proves transport-up; only a bare
309
+ # 200, or a same-authority launch-URL exchange (303 + cookie-authenticated 200)
310
+ # proves application readiness. A naked 401 therefore never counts as ready.
311
+ ready_probe() {
312
+ local status url jar exchange authenticated
313
+ current_owned_listener || return 1
314
+ # The test fixture records the proven runtime identity as soon as ownership
315
+ # is established. This is not production discovery or port-based cleanup:
316
+ # teardown later signals only this immutable PID/start-token lease.
317
+ test_register_pid "$current_listener_pid" instance-listener
318
+ status=$(http_status)
319
+ current_ownership_matches || return 1
320
+ if [ -n "$status" ] && [ "$status" != "000" ] && [ "$status" != "$last_transport_status" ]; then
321
+ last_transport_status=$status
322
+ cutover_event_required transport "$status"
323
+ if [ "$status" != "200" ]; then
324
+ wd_log "transport up on :$PORT (HTTP $status); application readiness still pending"
325
+ fi
326
+ fi
327
+ if [ "$status" = "200" ]; then
328
+ protected_ready=0
329
+ readiness_detail="plain HTTP 200"
330
+ return 0
331
+ fi
332
+
333
+ if [ -z "$launch_url_value" ]; then
334
+ launch_url_value=$(launch_url_from_output)
335
+ [ -n "$launch_url_value" ] && redact_launch_urls_in_output
336
+ fi
337
+ url=$launch_url_value
338
+ [ -n "$url" ] || return 1
339
+ if [ "$launch_url_reported" = "0" ]; then
340
+ wd_log "observed same-authority launch URL (credential redacted)"
341
+ cutover_event_required launch-url
342
+ launch_url_reported=1
343
+ fi
344
+ jar=$(mktemp "${TMPDIR:-/tmp}/ankh-guard-handoff.XXXXXX") || return 1
345
+ handoff_cookie_jar=$jar
346
+ chmod 600 "$jar" 2>/dev/null || true
347
+ exchange=$(curl -sS --noproxy '*' -c "$jar" -o /dev/null -w '%{http_code}' --max-time 3 "$url" 2>/dev/null || true)
348
+ cutover_event_required auth-exchange "${exchange:-0}"
349
+ if [ "$exchange" != "303" ]; then rm -f "$jar"; handoff_cookie_jar=''; return 1; fi
350
+ authenticated=$(curl -sS --noproxy '*' -b "$jar" -o /dev/null -w '%{http_code}' --max-time 3 "http://127.0.0.1:$PORT/" 2>/dev/null || true)
351
+ rm -f "$jar"
352
+ handoff_cookie_jar=''
353
+ cutover_event_required authenticated "${authenticated:-0}"
354
+ [ "$authenticated" = "200" ] || return 1
355
+ current_ownership_matches || return 1
356
+ protected_ready=1
357
+ readiness_detail="authenticated launch URL: 303 exchange, cookie / = 200"
358
+ return 0
359
+ }
360
+
361
+ # Whether the previous, proven listener armed an original tab for this
362
+ # transaction. All registered tabs are eligible to recover; the file contains
363
+ # only capability digests, never raw capabilities or bearer launch URLs.
364
+ browser_original_registered() {
365
+ node -e '
366
+ const fs = require("fs")
367
+ try {
368
+ const value = JSON.parse(fs.readFileSync(process.argv[1], "utf8"))
369
+ if (value.version !== 2 || value.cutoverId !== process.argv[2]
370
+ || !Array.isArray(value.registrations) || value.registrations.length < 1
371
+ || value.registrations.length > 64) process.exit(1)
372
+ let expectedPort = false
373
+ for (const registration of value.registrations) {
374
+ const authority = new URL(`http://${registration.authority ?? ""}`)
375
+ if (typeof registration.authority !== "string" || registration.authority === ""
376
+ || authority.host !== registration.authority
377
+ || authority.username !== "" || authority.password !== ""
378
+ || !/^[a-f0-9]{64}$/.test(registration.capabilitySha256)
379
+ || !Number.isFinite(registration.armedAt)) process.exit(1)
380
+ if (authority.port === process.argv[3]) expectedPort = true
381
+ }
382
+ if (!expectedPort) process.exit(1)
383
+ } catch { process.exit(1) }
384
+ ' "$BROWSER_HANDOFF_REQUEST_FILE" "$CUTOVER_ID" "$PORT" >/dev/null 2>&1
385
+ }
386
+
387
+ # Print non-secret acknowledgement evidence only when it names this exact
388
+ # stable listener identity and cutover role.
389
+ browser_ack_evidence() {
390
+ node -e '
391
+ const fs = require("fs")
392
+ try {
393
+ const value = JSON.parse(fs.readFileSync(process.argv[1], "utf8"))
394
+ const authority = new URL(`http://${value.authority ?? ""}`)
395
+ if (value.version !== 1 || value.cutoverId !== process.argv[2]
396
+ || value.role !== process.argv[3] || String(value.listenerPid) !== process.argv[4]
397
+ || value.listenerStartToken !== process.argv[5]
398
+ || !["original-tab", "fallback-tab"].includes(value.channel)
399
+ || !["existing-cookie", "launch-url"].includes(value.authentication)
400
+ || typeof value.authority !== "string" || value.authority === ""
401
+ || authority.host !== value.authority || authority.port !== process.argv[6]
402
+ || authority.username !== "" || authority.password !== ""
403
+ || !Number.isFinite(value.acknowledgedAt) || value.acknowledgedAt <= 0
404
+ || value.authority.includes("|")) process.exit(1)
405
+ process.stdout.write(`${value.channel}|${value.authentication}|${value.authority}`)
406
+ } catch { process.exit(1) }
407
+ ' "$BROWSER_HANDOFF_ACK_FILE" "$CUTOVER_ID" "$CUTOVER_ROLE" \
408
+ "$current_listener_pid" "$current_listener_start" "$PORT" 2>/dev/null
409
+ }
410
+
411
+ wait_for_browser_ack() {
412
+ local deadline evidence rest
413
+ case "$BROWSER_HANDOFF_TIMEOUT_SECONDS" in ''|*[!0-9]*) return 1 ;; esac
414
+ [ "$BROWSER_HANDOFF_TIMEOUT_SECONDS" -gt 0 ] || return 1
415
+ deadline=$(( $(now_ms) + BROWSER_HANDOFF_TIMEOUT_SECONDS * 1000 ))
416
+ while [ "$(now_ms)" -lt "$deadline" ]; do
417
+ [ -z "$(cutover_control_action)" ] || return 1
418
+ current_ownership_matches || return 1
419
+ evidence=$(browser_ack_evidence) || evidence=''
420
+ if [ -n "$evidence" ]; then
421
+ browser_ack_channel=${evidence%%|*}
422
+ rest=${evidence#*|}
423
+ browser_ack_authentication=${rest%%|*}
424
+ browser_ack_authority=${rest#*|}
425
+ return 0
426
+ fi
427
+ wd_sleep 0.25
428
+ done
429
+ return 1
430
+ }
431
+
432
+ # Browser handoff is deliberately after the ownership stability window and
433
+ # canary. An opener's exit status proves only that a fallback was attempted;
434
+ # readiness requires a page-authored acknowledgement naming the stable listener.
435
+ complete_browser_handoff() {
436
+ local fallback_url
437
+ current_ownership_matches || return 1
438
+ [ "$protected_ready" = "1" ] || return 0
439
+ if [ "$BROWSER_HANDOFF" = "required" ] && [ "$browser_handoff_done" = "0" ]; then
440
+ [ -n "$launch_url_value" ] || return 1
441
+ if browser_original_registered; then
442
+ wd_log "waiting for an armed original browser tab to acknowledge the final process"
443
+ if wait_for_browser_ack; then
444
+ browser_handoff_done=1
445
+ browser_handoff_reported=1
446
+ cutover_event_required browser-handoff acknowledged "$browser_ack_channel" \
447
+ "$browser_ack_authentication" "$browser_ack_authority"
448
+ wd_log "original browser tab acknowledged handoff ($browser_ack_authentication; authority $browser_ack_authority); all registered responsive tabs remain eligible to recover"
449
+ else
450
+ wd_log "original browser tab did not acknowledge within ${BROWSER_HANDOFF_TIMEOUT_SECONDS}s; falling back to system open"
451
+ fi
452
+ else
453
+ wd_log "no original browser tab registered before shutdown; falling back to system open"
454
+ fi
455
+ if [ "$browser_handoff_done" = "0" ]; then
456
+ fallback_url="${launch_url_value}#ankh-guard-handoff=${CUTOVER_ID}"
457
+ if open_launch_url "$fallback_url"; then
458
+ cutover_event_required browser-fallback-opened
459
+ wd_log "browser fallback open requested; waiting for page acknowledgement"
460
+ if wait_for_browser_ack; then
461
+ browser_handoff_done=1
462
+ browser_handoff_reported=1
463
+ cutover_event_required browser-handoff acknowledged "$browser_ack_channel" \
464
+ "$browser_ack_authentication" "$browser_ack_authority"
465
+ wd_log "browser acknowledged fallback handoff ($browser_ack_channel; authority $browser_ack_authority)"
466
+ fi
467
+ fi
468
+ fi
469
+ if [ "$browser_handoff_done" = "0" ]; then
470
+ if [ "$browser_handoff_reported" = "0" ]; then
471
+ wd_log "browser handoff failed: no page acknowledgement — not ready"
472
+ cutover_event_required browser-handoff failed
473
+ browser_handoff_reported=1
474
+ fi
475
+ return 1
476
+ fi
477
+ elif [ "$BROWSER_HANDOFF" = "off" ] && [ "$browser_handoff_reported" = "0" ]; then
478
+ cutover_event_required browser-handoff off
479
+ browser_handoff_reported=1
480
+ fi
481
+ current_ownership_matches || return 1
482
+ if [ "$browser_handoff_done" = "1" ]; then
483
+ readiness_detail="authenticated launch URL: 303 exchange, cookie / = 200, browser page acknowledged"
484
+ fi
485
+ return 0
486
+ }
487
+
488
+ # Initial HTTP/auth readiness is provisional. Hold the exact child PID/start
489
+ # identity and exact listener PID/start identity unchanged for a stability
490
+ # window, with no retry tolerated inside that window, before canary/terminal
491
+ # receipt. A child that exits after first returning 200 therefore fails the
492
+ # cutover instead of borrowing another process's response.
493
+ prove_stable_readiness() {
494
+ local expected_child=$child expected_child_start=$child_start_token
495
+ local expected_listener=$current_listener_pid expected_listener_start=$current_listener_start
496
+ local deadline
497
+ case "$READY_STABILITY_SECONDS" in ''|*[!0-9]*) return 1 ;; esac
498
+ [ "$READY_STABILITY_SECONDS" -gt 0 ] || return 1
499
+ deadline=$(( $(now_ms) + READY_STABILITY_SECONDS * 1000 ))
500
+ while [ "$(now_ms)" -lt "$deadline" ]; do
501
+ [ -z "$(cutover_control_action)" ] || return 1
502
+ [ "$child" = "$expected_child" ] && [ "$child_start_token" = "$expected_child_start" ] || return 1
503
+ current_listener_pid=$expected_listener
504
+ current_listener_start=$expected_listener_start
505
+ current_ownership_matches || return 1
506
+ ready_probe || return 1
507
+ [ "$current_listener_pid" = "$expected_listener" ] \
508
+ && [ "$current_listener_start" = "$expected_listener_start" ] || return 1
509
+ wd_sleep 0.25
510
+ done
511
+ current_listener_pid=$expected_listener
512
+ current_listener_start=$expected_listener_start
513
+ current_ownership_matches || return 1
514
+ readiness_detail="$readiness_detail; ownership stable ${READY_STABILITY_SECONDS}s (child $child, listener $current_listener_pid, retry 0)"
515
+ cutover_event_required ownership-stable "$CUTOVER_ROLE" "$child" "$child_start_token" \
516
+ "$current_listener_pid" "$current_listener_start" "$((READY_STABILITY_SECONDS * 1000))" 0
517
+ return 0
518
+ }
519
+
520
+ transport_up() {
521
+ local status
522
+ status=$(http_status)
523
+ [ -n "$status" ] && [ "$status" != "000" ]
524
+ }
525
+
526
+ process_start_token() {
527
+ if [ "$(uname -s 2>/dev/null)" = "Darwin" ] && [ -n "$PYTHON_BIN" ]; then
528
+ "$PYTHON_BIN" -c 'import ctypes,struct,sys;p=int(sys.argv[1]);b=ctypes.create_string_buffer(136);n=ctypes.CDLL("/usr/lib/libproc.dylib").proc_pidinfo(p,3,0,b,136);n == 136 or sys.exit(1);s,u=struct.unpack_from("QQ",b.raw,120);print(f"darwin:{s}:{u}",end="")' "$1" 2>/dev/null && return 0
529
+ fi
530
+ PROCESS_PS="$PS_BIN" PROCESS_SYSCTL="$SYSCTL_BIN" node -e '
531
+ const { createHash } = require("crypto")
532
+ const { existsSync, readFileSync } = require("fs")
533
+ const { execFileSync } = require("child_process")
534
+ const pid = Number(process.argv[1])
535
+ try {
536
+ const run = (file, args) => execFileSync(file, args, { encoding: "utf8", stdio: "pipe" }).trim()
537
+ const status = run(process.env.PROCESS_PS, ["-o", "stat=", "-p", String(pid)])
538
+ if (!status || status.startsWith("Z")) process.exit(1)
539
+ const procStat = `/proc/${pid}/stat`
540
+ if (existsSync(procStat)) {
541
+ const stat = readFileSync(procStat, "utf8")
542
+ const end = stat.lastIndexOf(")")
543
+ const fields = end < 0 ? [] : stat.slice(end + 1).trim().split(/\s+/)
544
+ const ticks = fields[19]
545
+ if (!ticks) process.exit(1)
546
+ let boot = "unknown-boot"
547
+ try { boot = readFileSync("/proc/sys/kernel/random/boot_id", "utf8").trim() || boot } catch {}
548
+ process.stdout.write(`linux:${boot}:${ticks}`)
549
+ process.exit(0)
550
+ }
551
+ const base = run(process.env.PROCESS_PS, ["-o", "sess=", "-o", "uid=", "-o", "lstart=", "-o", "command=", "-p", String(pid)])
552
+ if (!base) process.exit(1)
553
+ let boot = "unknown-boot"
554
+ try { if (process.env.PROCESS_SYSCTL) boot = run(process.env.PROCESS_SYSCTL, ["-n", "kern.boottime"]) || boot } catch {}
555
+ process.stdout.write("posix:" + createHash("sha256").update(boot).update("\0").update(base).digest("hex"))
556
+ } catch { process.exit(1) }
557
+ ' "$1" 2>/dev/null
558
+ }
559
+
560
+ # Test-only birth registration and event sink. The shipped watchdog is inert
561
+ # unless a fixture provides an explicit private run directory, token, and the
562
+ # built registrar path. Each shell/Node child initiates its own record so the
563
+ # outer Vitest process never has to discover a replacement through a mutable
564
+ # pidfile. Registration failure is diagnostic-only and cannot alter production
565
+ # supervision behavior.
566
+ test_register_pid() {
567
+ [ -n "${ANKH_GUARD_TEST_RUN_DIR:-}" ] || return 0
568
+ [ -n "${ANKH_GUARD_TEST_RUN_TOKEN:-}" ] || return 0
569
+ [ -f "${ANKH_GUARD_TEST_REGISTER_BIN:-}" ] || return 0
570
+ ANKH_GUARD_TEST_PROCESS_ROLE="$2" node "$ANKH_GUARD_TEST_REGISTER_BIN" register-pid "$1" "$2" >/dev/null 2>&1 || true
571
+ }
572
+
573
+ test_register_self() {
574
+ [ -n "${ANKH_GUARD_TEST_RUN_DIR:-}" ] || return 0
575
+ [ -n "${ANKH_GUARD_TEST_RUN_TOKEN:-}" ] || return 0
576
+ [ -f "${ANKH_GUARD_TEST_REGISTER_BIN:-}" ] || return 0
577
+ ANKH_GUARD_TEST_PROCESS_ROLE="$1" node "$ANKH_GUARD_TEST_REGISTER_BIN" register-parent "$1" >/dev/null 2>&1 || true
578
+ }
579
+
580
+ test_event_pid() {
581
+ [ -n "${ANKH_GUARD_TEST_RUN_DIR:-}" ] || return 0
582
+ [ -n "${ANKH_GUARD_TEST_RUN_TOKEN:-}" ] || return 0
583
+ [ -f "${ANKH_GUARD_TEST_REGISTER_BIN:-}" ] || return 0
584
+ ANKH_GUARD_TEST_PROCESS_ROLE="$2" node "$ANKH_GUARD_TEST_REGISTER_BIN" event-pid "$1" "$2" "$3" >/dev/null 2>&1 || true
585
+ }
586
+
587
+ test_event_self() {
588
+ [ -n "${ANKH_GUARD_TEST_RUN_DIR:-}" ] || return 0
589
+ [ -n "${ANKH_GUARD_TEST_RUN_TOKEN:-}" ] || return 0
590
+ [ -f "${ANKH_GUARD_TEST_REGISTER_BIN:-}" ] || return 0
591
+ ANKH_GUARD_TEST_PROCESS_ROLE="$1" node "$ANKH_GUARD_TEST_REGISTER_BIN" event-parent "$1" "$2" >/dev/null 2>&1 || true
592
+ }
593
+
594
+ now_ms() {
595
+ node -e 'process.stdout.write(String(Date.now()))'
596
+ }
597
+
598
+ identity_matches() {
599
+ local pid=$1 expected=$2 actual
600
+ [ -n "$pid" ] && [ -n "$expected" ] || return 1
601
+ actual=$(process_start_token "$pid")
602
+ [ -n "$actual" ] && [ "$actual" = "$expected" ]
603
+ }
604
+
605
+ listener_pids() {
606
+ "$LSOF_BIN" -tiTCP:"$PORT" -sTCP:LISTEN -P 2>/dev/null | sort -u
84
607
  }
85
608
 
86
- healthy() {
87
- code=$(curl -s --noproxy '*' -o /dev/null -w '%{http_code}' --max-time 3 "http://127.0.0.1:$PORT/")
88
- [ "$code" = "200" ]
609
+ pid_is_listener() {
610
+ local expected=$1 candidate
611
+ for candidate in $(listener_pids); do [ "$candidate" = "$expected" ] && return 0; done
612
+ return 1
613
+ }
614
+
615
+ pid_belongs_to_tree() {
616
+ local root=$1 cursor=$2 parent hops=0
617
+ while [ "$cursor" -gt 0 ] 2>/dev/null && [ "$hops" -lt 256 ]; do
618
+ [ "$cursor" = "$root" ] && return 0
619
+ parent=$("$PS_BIN" -o ppid= -p "$cursor" 2>/dev/null | tr -d ' ')
620
+ [ -n "$parent" ] || return 1
621
+ cursor=$parent
622
+ hops=$((hops + 1))
623
+ done
624
+ return 1
625
+ }
626
+
627
+ # Prove one unchanged listener inside the launched child tree. The response
628
+ # probe is accepted only while this identity remains true before and after
629
+ # the HTTP exchange, preventing a stale/foreign server on the same port from
630
+ # being mistaken for the target.
631
+ current_owned_listener() {
632
+ local pids count listener token
633
+ identity_matches "$child" "$child_start_token" || return 1
634
+ pids=$(listener_pids)
635
+ count=$(printf '%s\n' "$pids" | sed '/^$/d' | wc -l | tr -d ' ')
636
+ [ "$count" = "1" ] || return 1
637
+ listener=$(printf '%s\n' "$pids" | head -1)
638
+ pid_belongs_to_tree "$child" "$listener" || return 1
639
+ token=$(process_start_token "$listener")
640
+ [ -n "$token" ] || return 1
641
+ current_listener_pid=$listener
642
+ current_listener_start=$token
643
+ return 0
644
+ }
645
+
646
+ current_ownership_matches() {
647
+ local pids count token
648
+ identity_matches "$child" "$child_start_token" || return 1
649
+ pids=$(listener_pids)
650
+ count=$(printf '%s\n' "$pids" | sed '/^$/d' | wc -l | tr -d ' ')
651
+ [ "$count" = "1" ] || return 1
652
+ [ "$(printf '%s\n' "$pids" | head -1)" = "$current_listener_pid" ] || return 1
653
+ token=$(process_start_token "$current_listener_pid")
654
+ [ -n "$token" ] && [ "$token" = "$current_listener_start" ] \
655
+ && pid_belongs_to_tree "$child" "$current_listener_pid"
89
656
  }
90
657
 
91
658
  # Reap a pid AND its descendants, deepest first (best effort). The watchdog
92
659
  # guarantees the direct child; the sweep keeps grandchildren from outliving
93
660
  # the instance — a single-pid kill is what orphaned listeners and left the
94
661
  # EADDRINUSE race behind.
95
- kill_tree() {
96
- local pid=$1 sig=${2:-TERM} child
97
- for child in $(pgrep -P "$pid" 2>/dev/null); do
98
- kill_tree "$child" "$sig"
662
+ kill_frozen_tree() {
663
+ local pid=$1 sig=${2:-TERM} child parent unsafe=0
664
+ for child in $("$PGREP_BIN" -P "$pid" 2>/dev/null); do
665
+ kill -STOP "$child" 2>/dev/null || continue
666
+ parent=$("$PS_BIN" -o ppid= -p "$child" 2>/dev/null | tr -d ' ')
667
+ if [ "$parent" != "$pid" ]; then
668
+ kill -CONT "$child" 2>/dev/null || true
669
+ unsafe=1
670
+ continue
671
+ fi
672
+ kill_frozen_tree "$child" "$sig" || unsafe=1
99
673
  done
100
674
  kill -s "$sig" "$pid" 2>/dev/null || true
675
+ if [ "$sig" != "KILL" ]; then kill -CONT "$pid" 2>/dev/null || true; fi
676
+ [ "$unsafe" = "0" ]
677
+ }
678
+
679
+ kill_tree() {
680
+ local pid=$1 sig=${2:-TERM}
681
+ # Freeze before enumeration so a wrapper cannot exit and orphan its
682
+ # listener to PID 1 between pgrep and signal delivery.
683
+ kill -STOP "$pid" 2>/dev/null || return 0
684
+ kill_frozen_tree "$pid" "$sig"
685
+ }
686
+
687
+ stop_matching_identity() {
688
+ local pid=$1 token=$2 sig=${3:-TERM} actual
689
+ [ -n "$pid" ] && [ -n "$token" ] || return 2
690
+ # Authorization is checked while the PID is frozen. A pre-STOP check leaves
691
+ # a reuse window in which the signal can hit a new, unrelated process.
692
+ kill -STOP "$pid" 2>/dev/null || return 0
693
+ actual=$(process_start_token "$pid")
694
+ if [ -z "$actual" ] || [ "$actual" != "$token" ]; then
695
+ kill -CONT "$pid" 2>/dev/null || true
696
+ wd_log "refused signal $sig to pid $pid: frozen start identity did not match" >&2
697
+ return 2
698
+ fi
699
+ kill_frozen_tree "$pid" "$sig"
700
+ }
701
+
702
+ # Stop only the process identities captured while the old supervisor still
703
+ # owned them. The listener is also retained because a shell wrapper can exit
704
+ # and orphan its server between tree enumeration and signal delivery. Never
705
+ # replace this with "kill whatever owns the port" during a cutover.
706
+ stop_previous_owned_tree() {
707
+ local deadline pids foreign=0 identity_error=0
708
+ [ -n "$PREVIOUS_CHILD_PID" ] && [ -n "$PREVIOUS_CHILD_START" ] \
709
+ && [ -n "$PREVIOUS_LISTENER_PID" ] && [ -n "$PREVIOUS_LISTENER_START" ] || {
710
+ wd_log "cutover ownership proof is incomplete; refusing to stop by port" >&2
711
+ return 1
712
+ }
713
+ stop_matching_identity "$PREVIOUS_CHILD_PID" "$PREVIOUS_CHILD_START" TERM || identity_error=1
714
+ # The frozen root sweep normally signals the listener too. Give its exit
715
+ # teardown time to drop the socket before separately touching the captured
716
+ # listener PID; a dying process can retain a ps row after its executable
717
+ # identity is already gone.
718
+ wd_sleep 0.5
719
+ if pid_is_listener "$PREVIOUS_LISTENER_PID"; then
720
+ stop_matching_identity "$PREVIOUS_LISTENER_PID" "$PREVIOUS_LISTENER_START" TERM || identity_error=1
721
+ fi
722
+ deadline=$(( $(date +%s) + 15 ))
723
+ while [ "$(date +%s)" -lt "$deadline" ]; do
724
+ if ! identity_matches "$PREVIOUS_CHILD_PID" "$PREVIOUS_CHILD_START" \
725
+ && ! pid_is_listener "$PREVIOUS_LISTENER_PID"; then
726
+ break
727
+ fi
728
+ wd_sleep 0.2
729
+ done
730
+ if identity_matches "$PREVIOUS_CHILD_PID" "$PREVIOUS_CHILD_START"; then
731
+ stop_matching_identity "$PREVIOUS_CHILD_PID" "$PREVIOUS_CHILD_START" KILL || identity_error=1
732
+ fi
733
+ if pid_is_listener "$PREVIOUS_LISTENER_PID"; then
734
+ stop_matching_identity "$PREVIOUS_LISTENER_PID" "$PREVIOUS_LISTENER_START" KILL || identity_error=1
735
+ fi
736
+ wd_sleep 0.2
737
+ if identity_matches "$PREVIOUS_CHILD_PID" "$PREVIOUS_CHILD_START" \
738
+ || pid_is_listener "$PREVIOUS_LISTENER_PID"; then
739
+ wd_log "captured previous child/listener identity did not exit" >&2
740
+ return 1
741
+ fi
742
+ pids=$(listener_pids)
743
+ if [ -n "$pids" ]; then
744
+ wd_log "port :$PORT is owned by unapproved pid(s) $(printf '%s' "$pids" | paste -sd, -); refusing arbitrary cleanup" >&2
745
+ foreign=1
746
+ fi
747
+ [ "$foreign" = "0" ] && [ "$identity_error" = "0" ]
748
+ }
749
+
750
+ kill_current_owned_attempt() {
751
+ local identity_error=0
752
+ if [ -n "${child:-}" ] && [ -n "${child_start_token:-}" ]; then
753
+ stop_matching_identity "$child" "$child_start_token" TERM || identity_error=1
754
+ fi
755
+ wd_sleep 0.2
756
+ if [ -n "${current_listener_pid:-}" ] && [ -n "${current_listener_start:-}" ] \
757
+ && pid_is_listener "$current_listener_pid"; then
758
+ stop_matching_identity "$current_listener_pid" "$current_listener_start" TERM || identity_error=1
759
+ fi
760
+ wd_sleep 0.2
761
+ if [ -n "${child:-}" ] && [ -n "${child_start_token:-}" ] \
762
+ && identity_matches "$child" "$child_start_token"; then
763
+ stop_matching_identity "$child" "$child_start_token" KILL || identity_error=1
764
+ fi
765
+ if [ -n "${current_listener_pid:-}" ] && [ -n "${current_listener_start:-}" ] \
766
+ && pid_is_listener "$current_listener_pid"; then
767
+ stop_matching_identity "$current_listener_pid" "$current_listener_start" KILL || identity_error=1
768
+ fi
769
+ [ "$identity_error" = "0" ]
770
+ }
771
+
772
+ select_previous_spec() {
773
+ local reason=$1
774
+ [ -n "$PREVIOUS_START" ] && [ -n "$PREVIOUS_HOME" ] && [ -n "$PREVIOUS_REPO" ] \
775
+ && [ -n "$PREVIOUS_HARNESS_ROOT" ] || return 1
776
+ if [ -n "$TRANSITION_PLAN_SHA256" ]; then
777
+ if ! guard_cmd transition-rollback "$CUTOVER_ID" --state-dir "$STATE_DIR"; then
778
+ wd_log "filesystem transition rollback failed — refusing to start previous over target state" >&2
779
+ return 1
780
+ fi
781
+ transition_rolled_back=1
782
+ wd_log "filesystem transition rolled back; rejected target output retained in cutover quarantine"
783
+ fi
784
+ cutover_event_required restoring "$reason"
785
+ START_CMD="$PREVIOUS_START"
786
+ DSH_ROOT="$PREVIOUS_HOME"
787
+ REPO="$PREVIOUS_REPO"
788
+ HARNESS_ROOT="$PREVIOUS_HARNESS_ROOT"
789
+ PROFILE="${PREVIOUS_PROFILE:-web}"
790
+ export DSH_HOME="$DSH_ROOT"
791
+ CUTOVER_ROLE="previous"
792
+ browser_handoff_done=0
793
+ browser_handoff_reported=0
794
+ rm -f "$BROWSER_HANDOFF_ACK_FILE"
795
+ failures=0
796
+ reset_done=1
797
+ port_races=0
798
+ rm -f "$CONTROL_FILE" "$CONTROL_ABORT_FILE" "$CONTROL_RESTORE_FILE"
799
+ return 0
800
+ }
801
+
802
+ # Return 0 when a durable operator request was consumed; control_result tells
803
+ # the caller whether to launch previous or park according to wait-for-user.
804
+ handle_cutover_control() {
805
+ local action effective
806
+ control_result=''
807
+ action=$(cutover_control_action)
808
+ [ -n "$action" ] || return 1
809
+ cutover_event_required control-requested "$action"
810
+ effective=$action
811
+ if [ "$action" = "abort" ]; then effective=$CUTOVER_POLICY; fi
812
+ if [ "$effective" = "restore-previous" ]; then
813
+ if ! kill_current_owned_attempt; then
814
+ rm -f "$CONTROL_FILE" "$CONTROL_ABORT_FILE" "$CONTROL_RESTORE_FILE"
815
+ cutover_event_required awaiting-user "operator requested $action, but the frozen current identity no longer matched; refusing to signal an unapproved process"
816
+ control_result='wait'
817
+ return 0
818
+ fi
819
+ if [ -n "${child:-}" ]; then wait "$child" 2>/dev/null || true; fi
820
+ if select_previous_spec "operator requested $action; restoring previous complete launch specification"; then
821
+ control_result='restore'
822
+ return 0
823
+ fi
824
+ effective='wait-for-user'
825
+ fi
826
+ rm -f "$CONTROL_FILE" "$CONTROL_ABORT_FILE" "$CONTROL_RESTORE_FILE"
827
+ cutover_event_required awaiting-user "operator requested $action; recovery is waiting for user"
828
+ control_result='wait'
829
+ return 0
830
+ }
831
+
832
+ # Keep a recovered/otherwise healthy child available while a non-terminal
833
+ # cutover waits for explicit operator action. This path intentionally skips
834
+ # last-good stamping and continues to consume abort/restore requests.
835
+ wait_cutover_with_live_child() {
836
+ while identity_matches "$child" "$child_start_token"; do
837
+ if handle_cutover_control && [ "$control_result" = "restore" ]; then return 0; fi
838
+ wd_sleep 2
839
+ done
840
+ wait "$child" 2>/dev/null || true
101
841
  }
102
842
 
103
843
  # Echo a pid and all its descendants, one per line.
104
844
  pid_tree() {
105
845
  local pid=$1 child
106
846
  echo "$pid"
107
- for child in $(pgrep -P "$pid" 2>/dev/null); do
847
+ for child in $("$PGREP_BIN" -P "$pid" 2>/dev/null); do
108
848
  pid_tree "$child"
109
849
  done
110
850
  }
@@ -119,7 +859,7 @@ instance_listen_ports() {
119
859
  local pids
120
860
  pids=$(pid_tree "$1" | paste -sd, -)
121
861
  [ -n "$pids" ] || return 0
122
- lsof -nP -a -p "$pids" -iTCP -sTCP:LISTEN 2>/dev/null \
862
+ "$LSOF_BIN" -nP -a -p "$pids" -iTCP -sTCP:LISTEN 2>/dev/null \
123
863
  | awk 'NR > 1 { n = split($9, a, ":"); print a[n] }' | sort -u
124
864
  }
125
865
 
@@ -127,11 +867,11 @@ instance_listen_ports() {
127
867
  # listener (the one-time bounce that moves a running instance under supervision).
128
868
  free_port() {
129
869
  local pid p
130
- pid=$(lsof -tiTCP:$PORT -sTCP:LISTEN -P 2>/dev/null)
870
+ pid=$(listener_pids)
131
871
  if [ -n "$pid" ]; then
132
- echo "[watchdog] freeing :$PORT from pid(s) $pid"
872
+ wd_log "freeing :$PORT from pid(s) $pid"
133
873
  for p in $pid; do kill_tree "$p" TERM; done
134
- sleep 2
874
+ wd_sleep 2
135
875
  fi
136
876
  }
137
877
 
@@ -169,6 +909,60 @@ stamp_last_good_boot() {
169
909
  printf '{"revision":"%s","at":%s}\n' "$sha" "$(date +%s)000" > "$STATE_DIR/last-good-boot.json"
170
910
  }
171
911
 
912
+ # A healthy boot also proves the current PROFILE COMPOSITION runs: snapshot
913
+ # its inputs (the bundles patch layer + the profile manifest). This is the
914
+ # rollback target for failures a checkout reset cannot fix — a freshly
915
+ # installed plugin whose row breaks the real boot lives in the profile, not
916
+ # the repository.
917
+ snapshot_composition() {
918
+ local dir="$DSH_ROOT/profiles/$PROFILE"
919
+ [ -f "$dir/cordis.patch.yml" ] || return 0
920
+ mkdir -p "$STATE_DIR/last-good-composition"
921
+ cp "$dir/cordis.patch.yml" "$STATE_DIR/last-good-composition/"
922
+ if [ -f "$dir/package.json" ]; then cp "$dir/package.json" "$STATE_DIR/last-good-composition/"; fi
923
+ }
924
+
925
+ # Restore the snapshotted composition over the live one, backing the current
926
+ # (failing) inputs up first and computing the delta for the recovery report.
927
+ # Returns 1 when there is nothing to restore to (no snapshot, or the snapshot
928
+ # already IS the live composition — retrying a boot with unchanged inputs is
929
+ # pointless).
930
+ restore_composition() {
931
+ local snap="$STATE_DIR/last-good-composition"
932
+ local dir="$DSH_ROOT/profiles/$PROFILE"
933
+ [ -f "$snap/cordis.patch.yml" ] || return 1
934
+ local same=1
935
+ diff -q "$snap/cordis.patch.yml" "$dir/cordis.patch.yml" >/dev/null 2>&1 || same=0
936
+ if [ -f "$snap/package.json" ] || [ -f "$dir/package.json" ]; then
937
+ diff -q "$snap/package.json" "$dir/package.json" >/dev/null 2>&1 || same=0
938
+ fi
939
+ [ "$same" = "0" ] || return 1
940
+ # What the rollback unmounts — names for the recovery report.
941
+ comp_restore_detail=$(node -e '
942
+ const fs = require("fs")
943
+ const read = (f) => { try { return JSON.parse(fs.readFileSync(f, "utf8")) } catch { return {} } }
944
+ const live = read(process.argv[1]), snap = read(process.argv[2])
945
+ const added = (a, b) => a.filter((x) => !b.includes(x))
946
+ const parts = []
947
+ const rows = added(live.dsh?.profile?.bundles ?? [], snap.dsh?.profile?.bundles ?? [])
948
+ if (rows.length > 0) parts.push(`卸载挂载行: ${rows.join(", ")}`)
949
+ const deps = added(Object.keys(live.dependencies ?? {}), Object.keys(snap.dependencies ?? {}))
950
+ if (deps.length > 0) parts.push(`移除依赖: ${deps.join(", ")}`)
951
+ process.stdout.write(parts.join(";"))
952
+ ' "$dir/package.json" "$snap/package.json" 2>/dev/null)
953
+ if ! diff -q "$snap/cordis.patch.yml" "$dir/cordis.patch.yml" >/dev/null 2>&1; then
954
+ comp_restore_detail="${comp_restore_detail:+$comp_restore_detail;}回滚 profile patch 层变更"
955
+ fi
956
+ local backup="$STATE_DIR/composition-backup-$(date +%s)"
957
+ mkdir -p "$backup"
958
+ cp "$dir/cordis.patch.yml" "$backup/" 2>/dev/null || true
959
+ if [ -f "$dir/package.json" ]; then cp "$dir/package.json" "$backup/"; fi
960
+ cp "$snap/cordis.patch.yml" "$dir/cordis.patch.yml"
961
+ if [ -f "$snap/package.json" ]; then cp "$snap/package.json" "$dir/package.json"; fi
962
+ wd_log "restored the last healthy profile composition over $dir (failing inputs backed up to $backup)"
963
+ return 0
964
+ }
965
+
172
966
  # Roll back to a known-good revision — unless that revision already IS HEAD:
173
967
  # the reset would be a commit no-op whose only effect is wiping uncommitted
174
968
  # work (a real hazard with concurrent sessions on a shared checkout), so skip
@@ -178,10 +972,10 @@ rollback_to() {
178
972
  local sha="$1" head
179
973
  head=$(git -C "$REPO" rev-parse HEAD 2>/dev/null)
180
974
  if [ -n "$head" ] && [ "$sha" = "$head" ]; then
181
- echo "[watchdog] rollback target $sha is the current HEAD — skipping reset (nothing to roll back; a reset would only wipe uncommitted work)"
975
+ wd_log "rollback target $sha is the current HEAD — skipping reset (nothing to roll back; a reset would only wipe uncommitted work)"
182
976
  return 1
183
977
  fi
184
- echo "[watchdog] rolling repo back to last known-good $sha"
978
+ wd_log "rolling repo back to last known-good $sha"
185
979
  guard_reset "$sha"
186
980
  }
187
981
 
@@ -213,7 +1007,10 @@ guard_cmd() {
213
1007
  }
214
1008
 
215
1009
  guard_verify() {
216
- guard_cmd verify --repo "$REPO" --state-dir "$STATE_DIR" >/dev/null 2>&1
1010
+ # Revalidate the exact short-lived authorization selected before the stop.
1011
+ # A same-launch proof may outlive the original build credential's freshness,
1012
+ # but only while every fingerprinted deployment input remains identical.
1013
+ guard_cmd verify-restart --repo "$REPO" --state-dir "$STATE_DIR" >/dev/null 2>&1
217
1014
  }
218
1015
 
219
1016
  guard_reset() {
@@ -224,10 +1021,15 @@ guard_reset() {
224
1021
  # retry button that signals the watchdog (SIGUSR1) — no terminal needed.
225
1022
  page_script() {
226
1023
  cat <<'EOF'
1024
+ if (process.env.ANKH_GUARD_TEST_RUN_DIR && process.env.ANKH_GUARD_TEST_RUN_TOKEN
1025
+ && process.env.ANKH_GUARD_TEST_REGISTER_BIN) {
1026
+ require('child_process').spawnSync(process.execPath, [process.env.ANKH_GUARD_TEST_REGISTER_BIN,
1027
+ 'register-pid', String(process.pid), 'crash-page'], { stdio: 'ignore', env: process.env });
1028
+ }
227
1029
  const http = require('http');
228
1030
  const port = Number(process.env.WD_PORT || 3080);
229
1031
  const wd = Number(process.env.WD_PID);
230
- http.createServer((req, res) => {
1032
+ const handler = (req, res) => {
231
1033
  if (req.url === '/restart') {
232
1034
  try { process.kill(wd, 'SIGUSR1'); res.end('retrying...'); }
233
1035
  catch (e) { res.statusCode = 500; res.end('signal failed: ' + e.message); }
@@ -242,12 +1044,46 @@ http.createServer((req, res) => {
242
1044
  + '<div style="text-align:center"><h2>dsh 服务未能启动</h2>'
243
1045
  + '<p>看门狗多次尝试仍未拉起服务。点击重试,或查看看门狗日志。</p>'
244
1046
  + '<form action="/restart"><button style="font-size:18px;padding:10px 28px">重试</button></form></div></body>');
245
- }).listen(port, '127.0.0.1');
1047
+ };
1048
+ // An occupied port is the COMMON case at give-up (the boot failures were
1049
+ // often EADDRINUSE themselves). Dying on the bind error — an unhandled
1050
+ // 'error' event — would return the watchdog's `wait` and drop it back into
1051
+ // the boot loop, fighting the healthy occupant it just gave up against
1052
+ // (observed 2026-08-30 in an e2e rig: four give-up cycles, the occupant
1053
+ // killed over and over). Park instead: retry the bind every 5s; SIGUSR1
1054
+ // still re-arms the boot loop.
1055
+ let noted = false;
1056
+ function bind() {
1057
+ const server = http.createServer(handler);
1058
+ server.on('error', (e) => {
1059
+ if (e && e.code === 'EADDRINUSE') {
1060
+ if (!noted) {
1061
+ noted = true;
1062
+ process.stderr.write(new Date().toISOString() + ' [watchdog] crash page cannot bind :' + port + ' (occupied) — retrying every 5s; SIGUSR1 to ' + wd + ' re-arms the boot loop\n');
1063
+ }
1064
+ setTimeout(bind, 5000);
1065
+ return;
1066
+ }
1067
+ throw e;
1068
+ });
1069
+ server.listen(port, '127.0.0.1');
1070
+ }
1071
+
1072
+ bind();
246
1073
  EOF
247
1074
  }
248
1075
 
1076
+ park_cutover() {
1077
+ local reason=$1
1078
+ printf '%s launch cutover waiting: %s\n' "$(date '+%F %T')" "$reason" > "$GIVE_UP_MARKER"
1079
+ WD_PORT="$PORT" WD_PID="$$" node -e "$(page_script)" &
1080
+ page_pid=$!
1081
+ wait "$page_pid" 2>/dev/null || true
1082
+ page_pid=''
1083
+ }
1084
+
249
1085
  retry_on_usrs() {
250
- echo "[watchdog] USR1 received — clearing give-up marker and retrying"
1086
+ wd_log "USR1 received — clearing give-up marker and retrying"
251
1087
  rm -f "$GIVE_UP_MARKER"
252
1088
  if [ -n "${page_pid:-}" ]; then kill "$page_pid" 2>/dev/null; fi
253
1089
  failures=0
@@ -255,6 +1091,10 @@ retry_on_usrs() {
255
1091
  port_races=0
256
1092
  }
257
1093
 
1094
+ # Register before the first ownership branch can exit or detach more children.
1095
+ test_register_self "${ANKH_GUARD_TEST_PROCESS_ROLE:-watchdog}"
1096
+ test_event_self "${ANKH_GUARD_TEST_PROCESS_ROLE:-watchdog}" process-started
1097
+
258
1098
  # --supervise: one watchdog only. The claim must be atomic — a check-then-write
259
1099
  # (`[ -f ]` + `kill -0`, then `>`) is a TOCTOU window in which two watchdogs
260
1100
  # starting together both find no live owner, both write, and both supervise the
@@ -266,25 +1106,69 @@ if [ "$SUPERVISE" = "1" ]; then
266
1106
  mkdir -p "$(dirname "$PIDFILE")" 2>/dev/null || true
267
1107
  claimed=0
268
1108
  attempt=0
269
- while [ "$attempt" -lt 5 ]; do
270
- attempt=$((attempt + 1))
271
- if (set -C; echo $$ > "$PIDFILE") 2>/dev/null; then claimed=1; break; fi
1109
+ empty_reads=0
1110
+ if [ -n "${WD_TAKEOVER_FROM:-}" ]; then
272
1111
  owner=$(cat "$PIDFILE" 2>/dev/null)
273
- if [ -n "$owner" ] && kill -0 "$owner" 2>/dev/null; then
274
- echo "[watchdog] already supervised by pid $owner; exiting"
275
- exit 0
1112
+ if [ -z "${WD_TAKEOVER_FROM_START:-}" ] || [ "$owner" != "$WD_TAKEOVER_FROM" ] \
1113
+ || ! identity_matches "$WD_TAKEOVER_FROM" "$WD_TAKEOVER_FROM_START"; then
1114
+ wd_log "takeover refused: expected matching pidfile owner identity $WD_TAKEOVER_FROM, found ${owner:-none}" >&2
1115
+ exit 1
276
1116
  fi
277
- # Stale (owner gone) or empty pidfile: drop it and race for the claim
278
- # again. Losing that race is correct — the next pass sees a live owner and
279
- # exits through the branch above.
280
- rm -f "$PIDFILE"
281
- done
1117
+ takeover_tmp="$PIDFILE.takeover.$$"
1118
+ echo $$ > "$takeover_tmp"
1119
+ # Atomic rename is the ownership handoff commit point. The previous
1120
+ # watchdog already knows how to yield to a different live pidfile owner
1121
+ # without reaping its child, so this works even when that watchdog is the
1122
+ # older package version that had no reconfigure verb.
1123
+ mv -f "$takeover_tmp" "$PIDFILE"
1124
+ claimed=1
1125
+ wd_log "claimed supervision from watchdog $owner; old child remains running until the scheduled exit"
1126
+ supervisor_start_token=$(process_start_token "$$")
1127
+ [ -n "$supervisor_start_token" ] || { wd_log "could not capture replacement watchdog start identity" >&2; exit 1; }
1128
+ cutover_event_required supervisor-ready "$$" "$supervisor_start_token"
1129
+ else
1130
+ while [ "$attempt" -lt 5 ]; do
1131
+ attempt=$((attempt + 1))
1132
+ if (set -C; echo $$ > "$PIDFILE") 2>/dev/null; then claimed=1; break; fi
1133
+ owner=$(cat "$PIDFILE" 2>/dev/null)
1134
+ if [ -n "$owner" ]; then
1135
+ if kill -0 "$owner" 2>/dev/null; then
1136
+ wd_log "already supervised by pid $owner; exiting"
1137
+ exit 0
1138
+ fi
1139
+ # A real but dead claim: safe to drop below.
1140
+ empty_reads=0
1141
+ else
1142
+ # An EMPTY pidfile is a rival's claim mid-write: the noclobber create
1143
+ # and the echo are two disk operations, and a preempted winner sits
1144
+ # between them. Deleting the file here re-opens the race and can
1145
+ # cascade until every racer exhausts its attempts (observed under
1146
+ # deploy-gate load: 8 concurrent racers, zero survivors). Give the
1147
+ # writer a beat to land its pid; only treat the file as abandoned
1148
+ # after several consecutive empty reads.
1149
+ empty_reads=$((empty_reads + 1))
1150
+ if [ "$empty_reads" -le 3 ]; then
1151
+ attempt=$((attempt - 1))
1152
+ wd_sleep 0.2
1153
+ continue
1154
+ fi
1155
+ fi
1156
+ # Stale (owner gone, or abandoned mid-write): drop it and race for the
1157
+ # claim again. Losing that race is correct — the next pass sees a live
1158
+ # owner and exits through the branch above.
1159
+ rm -f "$PIDFILE"
1160
+ done
1161
+ fi
282
1162
  if [ "$claimed" != "1" ]; then
283
- echo "[watchdog] could not claim $PIDFILE after $attempt attempts" >&2
1163
+ wd_log "could not claim $PIDFILE after $attempt attempts" >&2
284
1164
  exit 1
285
1165
  fi
286
1166
  fi
287
1167
 
1168
+ if [ "$SUPERVISE" = "1" ] && [ "$claimed" = "1" ]; then
1169
+ test_event_self "${ANKH_GUARD_TEST_PROCESS_ROLE:-watchdog}" pidfile-published
1170
+ fi
1171
+
288
1172
  # Graceful launch: let the scheduling turn finish before the adoption bounce.
289
1173
  if [ "$DELAY" -gt 0 ] 2>/dev/null; then sleep "$DELAY"; fi
290
1174
 
@@ -292,7 +1176,14 @@ rm -f "$GIVE_UP_MARKER"
292
1176
  failures=0
293
1177
  reset_done=0
294
1178
  port_races=0
1179
+ target_attempt=0
1180
+ previous_attempt=0
295
1181
  yielded=0
1182
+ comp_restore_done=0
1183
+ comp_restored=0
1184
+ comp_restore_detail=''
1185
+ transition_applied=0
1186
+ transition_rolled_back=0
296
1187
 
297
1188
  trap 'retry_on_usrs' USR1
298
1189
 
@@ -302,43 +1193,228 @@ trap 'retry_on_usrs' USR1
302
1193
  # EADDRINUSE branch then had to free. SIGKILL cannot be trapped; the next
303
1194
  # start's free_port covers that case. (`set -u` — guard every var.)
304
1195
  cleanup() {
1196
+ if [ -n "${handoff_cookie_jar:-}" ]; then rm -f "$handoff_cookie_jar"; fi
305
1197
  if [ -n "${page_pid:-}" ]; then kill "$page_pid" 2>/dev/null; fi
306
1198
  # A YIELDING watchdog leaves its instance running for the new owner (the
307
1199
  # port is healthy; killing it would just make the successor respawn).
308
- if [ -n "${child:-}" ] && [ "${yielded:-0}" != "1" ]; then kill_tree "$child" TERM; fi
1200
+ if [ -n "${child:-}" ] && [ "${yielded:-0}" != "1" ]; then
1201
+ if [ -n "${CUTOVER_ID:-}" ] && [ -n "${child_start_token:-}" ]; then
1202
+ stop_matching_identity "$child" "$child_start_token" TERM || true
1203
+ else
1204
+ kill_tree "$child" TERM
1205
+ fi
1206
+ fi
309
1207
  # Drop the pidfile ONLY while it names us: a successor watchdog may have
310
1208
  # already claimed it in the restart window, and deleting theirs would let a
311
1209
  # second supervisor in.
312
1210
  if [ -f "$PIDFILE" ] && [ "$(cat "$PIDFILE" 2>/dev/null)" = "$$" ]; then
313
- rm -f "$PIDFILE"
1211
+ if [ -n "${WD_TAKEOVER_FROM:-}" ] && [ -n "${WD_TAKEOVER_FROM_START:-}" ] \
1212
+ && identity_matches "$WD_TAKEOVER_FROM" "$WD_TAKEOVER_FROM_START"; then
1213
+ takeover_restore="$PIDFILE.restore.$$"
1214
+ echo "$WD_TAKEOVER_FROM" > "$takeover_restore"
1215
+ mv -f "$takeover_restore" "$PIDFILE"
1216
+ wd_log "takeover aborted while old watchdog $WD_TAKEOVER_FROM is alive — restored its pidfile claim"
1217
+ else
1218
+ rm -f "$PIDFILE"
1219
+ fi
314
1220
  fi
315
1221
  return 0
316
1222
  }
317
1223
  trap cleanup EXIT
318
1224
  trap 'cleanup; exit 143' TERM INT
319
1225
 
320
- if [ "${WD_WAIT_OWNER:-0}" = "1" ]; then
1226
+ cutover_control_signal() {
1227
+ # The durable marker is the signal. Let the main loop freeze and reap the
1228
+ # authoritative tree; killing only the wrapper here recreates the orphan
1229
+ # listener race.
1230
+ if [ -n "${page_pid:-}" ]; then kill "$page_pid" 2>/dev/null || true; fi
1231
+ }
1232
+ trap 'cutover_control_signal' USR2
1233
+
1234
+ # Test readiness is stronger than pidfile publication: the control handler is
1235
+ # installed and the shell has yielded through one scheduler tick. Production
1236
+ # readiness remains unchanged; only explicit test event consumers observe it.
1237
+ test_event_self "${ANKH_GUARD_TEST_PROCESS_ROLE:-watchdog}" handler-installed
1238
+ if [ -n "${ANKH_GUARD_TEST_RUN_DIR:-}" ]; then wd_sleep 0.01; fi
1239
+ test_event_self "${ANKH_GUARD_TEST_PROCESS_ROLE:-watchdog}" keepalive-first-tick
1240
+ test_event_self "${ANKH_GUARD_TEST_PROCESS_ROLE:-watchdog}" ready
1241
+
1242
+ write_cutover_restart_marker() {
1243
+ node -e '
1244
+ const fs = require("fs")
1245
+ const [file, id, initiator] = process.argv.slice(1)
1246
+ fs.writeFileSync(file, JSON.stringify({
1247
+ reason: "launch configuration cutover",
1248
+ cutoverId: id,
1249
+ requestedAt: Date.now(),
1250
+ ...(initiator === "" ? {} : { initiator }),
1251
+ }) + "\n")
1252
+ ' "$RESTART_MARKER" "$CUTOVER_ID" "${WD_INITIATOR:-}"
1253
+ }
1254
+
1255
+ consume_takeover_control() {
1256
+ if handle_cutover_control; then
1257
+ if [ "$control_result" = "wait" ]; then
1258
+ wd_log "operator control parked cutover before the irreversible boundary; previous host remains available"
1259
+ park_cutover "operator abort follows wait-for-user policy"
1260
+ rm -f "$GIVE_UP_MARKER"
1261
+ else
1262
+ wd_log "operator control selected previous launch configuration before takeover"
1263
+ fi
1264
+ fi
1265
+ }
1266
+
1267
+ if [ -n "${WD_TAKEOVER_FROM:-}" ]; then
1268
+ # The atomic pidfile claim above is the cutover commit point. From here the
1269
+ # replacement watchdog — not the short-lived reconfigure caller — owns the
1270
+ # delayed old-child stop, so a caller/session death cannot strand the
1271
+ # transaction between "supervisor-ready" and "host stopped".
1272
+ cutover_delay="${WD_CUTOVER_DELAY_SECONDS:-5}"
1273
+ wd_log "supervision claimed; leaving the old host uninterrupted for ${cutover_delay}s"
1274
+ cutover_delay_deadline=$(node -e '
1275
+ const delay = Number(process.argv[1])
1276
+ if (!Number.isFinite(delay) || delay < 0) process.exit(1)
1277
+ process.stdout.write(String(Date.now() + delay * 1000))
1278
+ ' "$cutover_delay") || { wd_log "invalid WD_CUTOVER_DELAY_SECONDS" >&2; exit 1; }
1279
+ while [ "$(now_ms)" -lt "$cutover_delay_deadline" ]; do
1280
+ consume_takeover_control
1281
+ wd_sleep 0.2
1282
+ done
1283
+ # Do not publish the restart marker while the previous watchdog is still
1284
+ # alive. Older watchdogs consume that marker themselves; if one wins that
1285
+ # race, the replacement child can become healthy while the cutover receipt
1286
+ # remains permanently nonterminal. The previous child stays up throughout
1287
+ # this wait. A healthy old watchdog notices our pidfile claim on its next
1288
+ # supervision pass and yields without reaping the child.
1289
+ supervisor_yield_timeout_ms="${WD_SUPERVISOR_YIELD_TIMEOUT_MS:-15000}"
1290
+ case "$supervisor_yield_timeout_ms" in ''|*[!0-9]*) wd_log "invalid WD_SUPERVISOR_YIELD_TIMEOUT_MS" >&2; exit 1 ;; esac
1291
+ [ "$supervisor_yield_timeout_ms" -ge 100 ] || { wd_log "WD_SUPERVISOR_YIELD_TIMEOUT_MS must be at least 100" >&2; exit 1; }
1292
+ supervisor_yield_deadline=$(( $(now_ms) + supervisor_yield_timeout_ms ))
1293
+ supervisor_was_live=0
1294
+ supervisor_timed_out=0
1295
+ supervisor_retirement='identity-gone'
1296
+ if identity_matches "$WD_TAKEOVER_FROM" "$WD_TAKEOVER_FROM_START"; then
1297
+ supervisor_was_live=1
1298
+ wd_log "waiting up to ${supervisor_yield_timeout_ms}ms for old watchdog $WD_TAKEOVER_FROM to yield; old host remains available"
1299
+ fi
1300
+ while identity_matches "$WD_TAKEOVER_FROM" "$WD_TAKEOVER_FROM_START"; do
1301
+ consume_takeover_control
1302
+ if [ "$(now_ms)" -ge "$supervisor_yield_deadline" ]; then
1303
+ supervisor_timed_out=1
1304
+ wd_log "old watchdog $WD_TAKEOVER_FROM did not yield in ${supervisor_yield_timeout_ms}ms; retiring its frozen, revalidated identity"
1305
+ if stop_matching_identity "$WD_TAKEOVER_FROM" "$WD_TAKEOVER_FROM_START" TERM; then
1306
+ supervisor_retirement='forced'
1307
+ retirement_deadline=$(( $(date +%s) + 3 ))
1308
+ while identity_matches "$WD_TAKEOVER_FROM" "$WD_TAKEOVER_FROM_START" \
1309
+ && [ "$(date +%s)" -lt "$retirement_deadline" ]; do wd_sleep 0.2; done
1310
+ stop_matching_identity "$WD_TAKEOVER_FROM" "$WD_TAKEOVER_FROM_START" KILL || true
1311
+ else
1312
+ # The PID changed between the loop predicate and SIGSTOP. The helper
1313
+ # resumed it without delivering TERM; the authorized old identity is
1314
+ # gone, so proceed using only the separately captured child identities.
1315
+ supervisor_retirement='identity-gone'
1316
+ fi
1317
+ break
1318
+ fi
1319
+ wd_sleep 0.2
1320
+ done
1321
+ if [ "$supervisor_was_live" = "1" ] && [ "$supervisor_timed_out" = "0" ]; then
1322
+ supervisor_retirement='yielded'
1323
+ fi
1324
+ cutover_event_required previous-supervisor-retired "$supervisor_retirement"
1325
+ # Publish the intentional-restart marker only at the irreversible boundary.
1326
+ # Writing it in the reconfigure caller lets the OLD watchdog consume and
1327
+ # clear it before yielding, leaving the final child ready but the receipt
1328
+ # permanently nonterminal. The old instance still sees the marker during
1329
+ # SIGTERM and can snapshot interrupted sessions with the correct initiator.
1330
+ write_cutover_restart_marker
1331
+ if ! stop_previous_owned_tree; then
1332
+ WD_TAKEOVER_FROM=""
1333
+ cutover_event_required awaiting-user "could not stop the captured previous child/listener identity without touching an unapproved port owner"
1334
+ printf '%s launch cutover waiting: previous ownership could not be retired\n' "$(date '+%F %T')" > "$GIVE_UP_MARKER"
1335
+ WD_PORT="$PORT" WD_PID="$$" node -e "$(page_script)" &
1336
+ page_pid=$!
1337
+ wait "$page_pid"
1338
+ exit 1
1339
+ fi
1340
+ wd_log "captured previous child $PREVIOUS_CHILD_PID and listener $PREVIOUS_LISTENER_PID exited — taking over :$PORT"
1341
+ # This is the irreversible boundary. Never restore a possibly recycled old
1342
+ # supervisor pid during a much later cleanup.
1343
+ WD_TAKEOVER_FROM=""
1344
+ elif [ -n "$CUTOVER_ID" ]; then
1345
+ # OS-level crash recovery: this watchdog claimed a stale/empty pidfile and
1346
+ # resumes the atomically selected side of an existing transaction. Replace
1347
+ # any orphan listener from the failed supervisor, then prove a fresh final
1348
+ # child; never compact the transaction merely because its driver died.
1349
+ wd_log "resuming launch cutover $CUTOVER_ID on selected side $CUTOVER_ROLE"
1350
+ supervisor_start_token=$(process_start_token "$$")
1351
+ [ -n "$supervisor_start_token" ] || { wd_log "could not capture resumed watchdog start identity" >&2; exit 1; }
1352
+ cutover_event_required supervisor-ready "$$" "$supervisor_start_token"
1353
+ write_cutover_restart_marker
1354
+ if ! stop_previous_owned_tree; then
1355
+ cutover_event_required awaiting-user "resume could not prove the shared port free from the captured previous identity"
1356
+ printf '%s launch cutover waiting: port ownership is ambiguous\n' "$(date '+%F %T')" > "$GIVE_UP_MARKER"
1357
+ WD_PORT="$PORT" WD_PID="$$" node -e "$(page_script)" &
1358
+ page_pid=$!
1359
+ wait "$page_pid"
1360
+ exit 1
1361
+ fi
1362
+ elif [ "${WD_WAIT_OWNER:-0}" = "1" ]; then
321
1363
  # Adoption ahead of a self-restart: the current owner exits on its own.
322
- echo "[watchdog] waiting for the current owner of :$PORT to exit"
323
- while lsof -tiTCP:$PORT -sTCP:LISTEN -P >/dev/null 2>&1; do sleep 1; done
324
- echo "[watchdog] port free — taking over"
1364
+ wd_log "waiting for the current owner of :$PORT to exit"
1365
+ while "$LSOF_BIN" -tiTCP:"$PORT" -sTCP:LISTEN -P >/dev/null 2>&1; do wd_sleep 1; done
1366
+ wd_log "port free — taking over"
325
1367
  else
326
1368
  free_port
327
1369
  fi
328
1370
 
329
1371
  while true; do
330
- # Self-heal the ownership claim FIRST: if the state dir (or the pidfile) was
331
- # cleaned underneath a live watchdog, reclaim it; if another LIVE watchdog
332
- # now holds it, yield — two supervisors on one port reap each other's
333
- # instance (observed: stale watchdog + deleted pidfile → second watchdog
334
- # spawned → both fought over the port).
1372
+ # Self-heal the ownership claim before processing control or changing live
1373
+ # state. If another supervisor owns the pidfile, this process has no
1374
+ # authority to apply or roll back a filesystem transition.
335
1375
  if [ "$SUPERVISE" = "1" ]; then
336
1376
  if [ ! -f "$PIDFILE" ]; then (set -C; echo $$ > "$PIDFILE") 2>/dev/null || true; fi
337
1377
  pidowner=$(cat "$PIDFILE" 2>/dev/null)
338
1378
  if [ -n "$pidowner" ] && [ "$pidowner" != "$$" ] && kill -0 "$pidowner" 2>/dev/null; then
339
- echo "[watchdog] pidfile now owned by live pid $pidowner — yielding"
1379
+ wd_log "pidfile now owned by live pid $pidowner — yielding"
340
1380
  yielded=1
341
- exit 0
1381
+ # Non-zero keeps launchd/systemd's stable launcher alive: it restarts,
1382
+ # reads the newly selected durable spec, then waits behind the successor.
1383
+ # A detached parent simply observes the code and is unaffected.
1384
+ exit 75
1385
+ fi
1386
+ fi
1387
+ if [ -n "$CUTOVER_ID" ] && handle_cutover_control; then
1388
+ if [ "$control_result" = "wait" ]; then
1389
+ park_cutover "operator abort follows wait-for-user policy"
1390
+ continue
1391
+ fi
1392
+ wd_log "operator control selected the previous complete launch specification"
1393
+ fi
1394
+ if [ -n "$TRANSITION_PLAN_SHA256" ]; then
1395
+ if [ "$CUTOVER_ROLE" = "target" ] && [ "$transition_applied" = "0" ]; then
1396
+ if guard_cmd transition-apply "$CUTOVER_ID" --state-dir "$STATE_DIR"; then
1397
+ transition_applied=1
1398
+ wd_log "filesystem transition applied after previous stopped and before target start"
1399
+ else
1400
+ wd_log "filesystem transition apply failed — target will not start" >&2
1401
+ if [ "$CUTOVER_POLICY" = "restore-previous" ] \
1402
+ && select_previous_spec "filesystem transition failed before target start; restoring previous spec"; then
1403
+ continue
1404
+ fi
1405
+ cutover_event_required awaiting-user "filesystem transition failed; target was not started and previous was not restored"
1406
+ park_cutover "filesystem transition failed; inspect the cutover transition journal"
1407
+ continue
1408
+ fi
1409
+ elif [ "$CUTOVER_ROLE" = "previous" ] && [ "$transition_rolled_back" = "0" ]; then
1410
+ if guard_cmd transition-rollback "$CUTOVER_ID" --state-dir "$STATE_DIR"; then
1411
+ transition_rolled_back=1
1412
+ wd_log "filesystem transition rollback verified before previous start"
1413
+ else
1414
+ cutover_event_required awaiting-user "filesystem transition rollback failed; previous was not started"
1415
+ park_cutover "filesystem transition rollback failed; previous remains stopped"
1416
+ continue
1417
+ fi
342
1418
  fi
343
1419
  fi
344
1420
  # Snapshot BEFORE this boot rewrites it: the stamp exists iff this
@@ -348,7 +1424,16 @@ while true; do
348
1424
  # watchdog still judges correctly.
349
1425
  had_boot_stamp=0
350
1426
  [ -f "$STATE_DIR/last-good-boot.json" ] && had_boot_stamp=1
351
- echo "[watchdog] starting instance on :$PORT (failures=$failures)"
1427
+ if [ "$CUTOVER_ROLE" = "target" ]; then
1428
+ target_attempt=$((target_attempt + 1))
1429
+ current_attempt=$target_attempt
1430
+ else
1431
+ previous_attempt=$((previous_attempt + 1))
1432
+ current_attempt=$previous_attempt
1433
+ fi
1434
+ # Keep the long-standing "starting instance" prefix stable for operators and
1435
+ # log consumers; the role is additive cutover metadata.
1436
+ wd_log "starting instance on :$PORT (role=$CUTOVER_ROLE, failures=$failures, attempt=$current_attempt)"
352
1437
  # Capture this attempt's output for failure-domain classification. Plain
353
1438
  # redirection only — never > >(tee …) process substitution: a sandboxed or
354
1439
  # detached spawner can EPERM on the /dev/fd/N that >() opens (workspace-write
@@ -356,32 +1441,91 @@ while true; do
356
1441
  # log is mirrored into this log below; a healthy run's boot message names
357
1442
  # the file its output lives in.
358
1443
  : > "$ATTEMPT_LOG"
1444
+ chmod 600 "$ATTEMPT_LOG" 2>/dev/null || true
359
1445
  launch_instance > "$ATTEMPT_LOG" 2>&1 &
360
1446
  child=$!
361
- # Boot window: the instance is up when the port answers 200.
1447
+ child_start_token=$(process_start_token "$child")
1448
+ [ -n "$child_start_token" ] || child_start_token="unavailable-$child"
1449
+ cutover_event_required child-started "$CUTOVER_ROLE" "$current_attempt" "$child" "$child_start_token"
1450
+ last_transport_status=''
1451
+ launch_url_reported=0
1452
+ launch_url_value=''
1453
+ readiness_detail=''
1454
+ protected_ready=0
1455
+ # Launch URLs and browser cookies are process-bound. Every retry must prove
1456
+ # and hand off its own URL; an accepted URL from a rejected child is stale.
1457
+ browser_handoff_done=0
1458
+ browser_handoff_reported=0
1459
+ current_listener_pid=''
1460
+ current_listener_start=''
1461
+ # Boot window: transport-up is not enough. A protected root can answer 401;
1462
+ # ready_probe completes the process's announced launch-URL cookie exchange.
362
1463
  up=0
1464
+ readiness_failure_detail="readiness not proven within ${BOOT_TIMEOUT}s"
363
1465
  boot_limit=$(( $(date +%s) + BOOT_TIMEOUT ))
364
1466
  while [ "$(date +%s)" -lt "$boot_limit" ]; do
365
- if ! kill -0 "$child" 2>/dev/null; then break; fi
366
- if healthy; then up=1; break; fi
367
- sleep 1
1467
+ if [ -n "$(cutover_control_action)" ]; then
1468
+ readiness_failure_detail="operator control interrupted readiness"
1469
+ kill_current_owned_attempt
1470
+ break
1471
+ fi
1472
+ if ! kill -0 "$child" 2>/dev/null; then
1473
+ readiness_failure_detail="child exited before ownership-stable readiness"
1474
+ break
1475
+ fi
1476
+ if ready_probe; then
1477
+ if prove_stable_readiness; then
1478
+ up=1
1479
+ else
1480
+ readiness_failure_detail="provisional readiness did not retain one child/listener identity through the stability window"
1481
+ fi
1482
+ # Readiness that cannot hold the same child/listener identity for the
1483
+ # stability window is an attempt failure, not an invitation to attach
1484
+ # to whichever process next answers on the shared port.
1485
+ break
1486
+ fi
1487
+ wd_sleep 1
368
1488
  done
369
1489
 
370
1490
  if [ "$up" = "0" ]; then
371
1491
  # Read the bound ports BEFORE reaping — once the child is gone there is no
372
1492
  # way left to tell "never started" from "started on the wrong port".
373
1493
  bound=""
374
- if kill -0 "$child" 2>/dev/null; then bound=$(instance_listen_ports "$child"); fi
1494
+ if identity_matches "$child" "$child_start_token"; then bound=$(instance_listen_ports "$child"); fi
375
1495
  # Never came up (or died); stop a still-alive child and reap it.
376
- if kill -0 "$child" 2>/dev/null; then kill "$child" 2>/dev/null; fi
1496
+ kill_current_owned_attempt
377
1497
  wait "$child" 2>/dev/null
1498
+ # Strip bearer launch URLs before any durable failure output is mirrored.
1499
+ redact_launch_urls_in_output
378
1500
  # Mirror the captured output into the watchdog log: with plain redirection
379
1501
  # (see the launch site) the attempt log is the only place the failure was
380
1502
  # written, and the watchdog log is where an operator looks first.
381
1503
  sed 's/^/[instance] /' "$ATTEMPT_LOG" 2>/dev/null
382
1504
 
1505
+ if [ -n "$CUTOVER_ID" ] && [ -n "$(cutover_control_action)" ]; then
1506
+ failures=$((failures + 1))
1507
+ cutover_event_required attempt-failed "$CUTOVER_ROLE" "$current_attempt" "operator control interrupted readiness"
1508
+ if handle_cutover_control; then
1509
+ if [ "$control_result" = "restore" ]; then
1510
+ wd_log "operator control interrupted target readiness — restoring previous"
1511
+ continue
1512
+ fi
1513
+ park_cutover "operator abort follows wait-for-user policy"
1514
+ continue
1515
+ fi
1516
+ fi
1517
+
383
1518
  if grep -q 'EADDRINUSE' "$ATTEMPT_LOG" 2>/dev/null; then
384
1519
  if grep 'EADDRINUSE' "$ATTEMPT_LOG" | grep -qE "[:.]$PORT([^0-9]|$)"; then
1520
+ if [ -n "$CUTOVER_ID" ]; then
1521
+ # The target never inherits authority to kill an arbitrary owner of
1522
+ # the shared port. Count the attempt so the approved full-spec
1523
+ # recovery policy runs; any listener previously proven inside this
1524
+ # attempt was already retired by kill_current_owned_attempt.
1525
+ wd_log "cutover $CUTOVER_ROLE hit EADDRINUSE on :$PORT — refusing port-based cleanup; counting a $CUTOVER_ROLE failure"
1526
+ readiness_failure_detail="EADDRINUSE on supervised :$PORT; cutover refused port-based cleanup"
1527
+ port_races=5
1528
+ else
385
1529
  # The supervised port was still held (a leftover process, a slow exit)
386
1530
  # — an operational race, not a code regression. The watchdog owns this
387
1531
  # port, so free it and retry WITHOUT counting toward rollback or
@@ -389,24 +1533,26 @@ while true; do
389
1533
  # is outside this watchdog's reach and retrying is a hot spin.
390
1534
  port_races=$((port_races + 1))
391
1535
  if [ "$port_races" -le 5 ]; then
392
- echo "[watchdog] boot hit EADDRINUSE on :$PORT — freeing the port and retrying (not a code failure, attempt $port_races/5)"
1536
+ wd_log "boot hit EADDRINUSE on :$PORT — freeing the port and retrying (not a code failure, attempt $port_races/5)"
393
1537
  free_port
394
1538
  continue
395
1539
  fi
396
- echo "[watchdog] :$PORT is still held after 5 free attempts — counting this as a boot failure"
1540
+ wd_log ":$PORT is still held after 5 free attempts — counting this as a boot failure"
1541
+ fi
397
1542
  else
398
1543
  # EADDRINUSE on a port this watchdog does not own: the start command
399
1544
  # targets somewhere else, and freeing :$PORT cannot release it. The
400
1545
  # unconditional retry this replaces never counted the attempt, so a
401
1546
  # start command aimed at an occupied foreign port respawned the
402
1547
  # instance in a tight loop with no backoff and no give-up.
403
- echo "[watchdog] boot hit EADDRINUSE on a port other than the supervised :$PORT — the --start command targets a port this watchdog does not own; freeing :$PORT cannot fix that"
1548
+ wd_log "boot hit EADDRINUSE on a port other than the supervised :$PORT — the --start command targets a port this watchdog does not own; freeing :$PORT cannot fix that"
404
1549
  reset_done=1
405
1550
  fi
406
1551
  fi
407
1552
 
408
1553
  failures=$((failures + 1))
409
- echo "[watchdog] instance failed to come up (failure #$failures)"
1554
+ wd_log "instance failed to come up (failure #$failures: $readiness_failure_detail)"
1555
+ cutover_event_required attempt-failed "$CUTOVER_ROLE" "$current_attempt" "$readiness_failure_detail"
410
1556
 
411
1557
  # The instance came up on a port this watchdog does not own: a start-command
412
1558
  # argument, not a code regression. Resetting the checkout cannot change a
@@ -414,18 +1560,53 @@ while true; do
414
1560
  # whose subject lives outside the repository) and keep counting toward the
415
1561
  # crash page, which is what makes the misconfiguration visible.
416
1562
  if [ -n "$bound" ] && ! printf '%s\n' "$bound" | grep -qx "$PORT"; then
417
- echo "[watchdog] instance bound :$(printf '%s' "$bound" | paste -sd, -) but supervision owns :$PORT — the --start command does not bind the supervised port; a repository rollback cannot fix that"
1563
+ wd_log "instance bound :$(printf '%s' "$bound" | paste -sd, -) but supervision owns :$PORT — the --start command does not bind the supervised port; a repository rollback cannot fix that"
418
1564
  reset_done=1
419
1565
  fi
420
1566
 
421
- if [ "$failures" -ge 2 ] && [ "$reset_done" -eq 0 ]; then
1567
+ # A launch cutover recovers the complete previous spec or waits, exactly as
1568
+ # approved before the stop. It never falls through to the ordinary
1569
+ # repository/composition reset machinery: neither can repair a command,
1570
+ # home, credential repo, host root, or profile change as one unit.
1571
+ if [ -n "$CUTOVER_ID" ] && [ "$failures" -ge "$TARGET_FAILURE_LIMIT" ]; then
1572
+ if [ "$CUTOVER_ROLE" = "target" ] && [ "$CUTOVER_POLICY" = "restore-previous" ] \
1573
+ && [ -n "$PREVIOUS_START" ] && [ -n "$PREVIOUS_HOME" ] && [ -n "$PREVIOUS_REPO" ] \
1574
+ && [ -n "$PREVIOUS_HARNESS_ROOT" ]; then
1575
+ wd_log "target launch failed after $failures attempt(s) — restoring the approved previous launch specification"
1576
+ if select_previous_spec "target failed after $failures attempt(s); restoring previous spec"; then
1577
+ continue
1578
+ fi
1579
+ cutover_event_required awaiting-user "target failed and filesystem transition rollback did not complete; previous was not started"
1580
+ park_cutover "target failed; previous restore is blocked by filesystem transition rollback"
1581
+ continue
1582
+ fi
1583
+ wd_log "launch cutover cannot become ready — approved policy is ${CUTOVER_POLICY:-wait-for-user}; parking for user action"
1584
+ cutover_event_required awaiting-user "$CUTOVER_ROLE launch failed after $failures attempt(s)"
1585
+ printf '%s launch cutover waiting after %s failures\n' "$(date '+%F %T')" "$failures" > "$GIVE_UP_MARKER"
1586
+ WD_PORT="$PORT" WD_PID="$$" node -e "$(page_script)" &
1587
+ page_pid=$!
1588
+ wait "$page_pid"
1589
+ page_pid=''
1590
+ continue
1591
+ fi
1592
+
1593
+ if [ -z "$CUTOVER_ID" ] && [ "$failures" -ge 2 ] && [ "$reset_done" -eq 0 ]; then
422
1594
  sha=$(rollback_sha)
423
1595
  if [ -z "$sha" ]; then
424
- echo "[watchdog] no guard credential/checkpoint recorded; cannot roll back"
1596
+ wd_log "no guard credential/checkpoint recorded; cannot roll back"
425
1597
  elif failure_subject_outside_repo; then
426
1598
  # The failure lives outside the checkout (profile overlay, installed
427
- # plugin, environment) — reverting the repository cannot fix it.
428
- echo "[watchdog] boot failure originates outside $REPO — repository rollback cannot fix it; leaving the checkout untouched"
1599
+ # plugin, environment) — reverting the repository cannot fix it. But
1600
+ # the COMPOSITION can be rolled back: if a healthy-boot snapshot of
1601
+ # the profile inputs exists and differs from the live one, restore it
1602
+ # (unmounting the newest plugin change) and retry with a clean count.
1603
+ if [ "$comp_restore_done" -eq 0 ] && restore_composition; then
1604
+ comp_restore_done=1
1605
+ comp_restored=1
1606
+ failures=0
1607
+ continue
1608
+ fi
1609
+ wd_log "boot failure originates outside $REPO — repository rollback cannot fix it; leaving the checkout untouched"
429
1610
  reset_done=1
430
1611
  else
431
1612
  if rollback_to "$sha"; then
@@ -440,36 +1621,157 @@ while true; do
440
1621
  fi
441
1622
 
442
1623
  if [ "$failures" -ge 4 ]; then
443
- echo "[watchdog] giving up after $failures consecutive failures"
1624
+ wd_log "giving up after $failures consecutive failures"
444
1625
  printf '%s giving up after %s failures\n' "$(date '+%F %T')" "$failures" > "$GIVE_UP_MARKER"
445
- echo "[watchdog] serving crash page on :$PORT — click 重试 or send SIGUSR1 to $$"
1626
+ wd_log "serving crash page on :$PORT — click 重试 or send SIGUSR1 to $$"
446
1627
  WD_PORT="$PORT" WD_PID="$$" node -e "$(page_script)" &
447
1628
  page_pid=$!
448
1629
  wait "$page_pid"
1630
+ page_pid=''
449
1631
  continue
450
1632
  fi
451
1633
 
452
- sleep $((failures * 5))
1634
+ wd_sleep $((failures * 5))
453
1635
  continue
454
1636
  fi
455
1637
 
456
1638
  # Instance is up.
457
- echo "[watchdog] instance up on :$PORT — instance output: $ATTEMPT_LOG"
458
- stamp_last_good_boot
1639
+ wd_log "instance ready on :$PORT ($readiness_detail) — instance output: $ATTEMPT_LOG"
1640
+ if [ "$comp_restored" = "1" ]; then
1641
+ # The boot only succeeded because the composition was rolled back — the
1642
+ # recovery (newest plugin change unmounted) must be reported, not silent.
1643
+ guard_cmd record-composition-recovery --state-dir "$STATE_DIR" --detail "$comp_restore_detail"
1644
+ comp_restored=0
1645
+ fi
459
1646
 
460
1647
  # Intentional restart: run the guard canary (credential fresh + HEAD match).
461
1648
  if [ -f "$RESTART_MARKER" ]; then
462
- if guard_verify; then
463
- echo "[watchdog] canary PASS — clearing restart marker"
1649
+ canary_proven=0
1650
+ if guard_verify && current_ownership_matches; then
1651
+ cutover_event_required canary "$CUTOVER_ROLE" pass
1652
+ # Persisting canary evidence can take long enough for a short-lived child
1653
+ # to exit. Recheck after the event and before the terminal ready event.
1654
+ if current_ownership_matches; then canary_proven=1; fi
1655
+ fi
1656
+ if [ "$canary_proven" = "1" ]; then
1657
+ if [ -n "$CUTOVER_ID" ] && ! complete_browser_handoff; then
1658
+ failures=$((failures + 1))
1659
+ cutover_event_required attempt-failed "$CUTOVER_ROLE" "$current_attempt" \
1660
+ "browser handoff failed after canary and stable ownership"
1661
+ if [ "$CUTOVER_ROLE" = "target" ] && [ "$CUTOVER_POLICY" = "restore-previous" ] \
1662
+ && [ -n "$PREVIOUS_START" ] && [ -n "$PREVIOUS_HOME" ] && [ -n "$PREVIOUS_REPO" ] \
1663
+ && [ -n "$PREVIOUS_HARNESS_ROOT" ]; then
1664
+ kill_current_owned_attempt
1665
+ wait "$child" 2>/dev/null || true
1666
+ if select_previous_spec "target canary passed but browser handoff failed; restoring previous spec"; then
1667
+ continue
1668
+ fi
1669
+ cutover_event_required awaiting-user "browser handoff failed and filesystem transition rollback did not complete; previous was not started"
1670
+ park_cutover "browser handoff failed; previous restore is blocked by filesystem transition rollback"
1671
+ continue
1672
+ fi
1673
+ if [ "$CUTOVER_ROLE" = "previous" ]; then
1674
+ rm -f "$RESTART_MARKER"
1675
+ cutover_event_required awaiting-user "restored previous host is ready, but browser handoff was not acknowledged"
1676
+ wait_cutover_with_live_child
1677
+ continue
1678
+ fi
1679
+ kill_current_owned_attempt
1680
+ wait "$child" 2>/dev/null || true
1681
+ rm -f "$RESTART_MARKER"
1682
+ cutover_event_required awaiting-user "target browser handoff failed after canary"
1683
+ printf '%s launch cutover waiting after browser handoff failure\n' "$(date '+%F %T')" > "$GIVE_UP_MARKER"
1684
+ WD_PORT="$PORT" WD_PID="$$" node -e "$(page_script)" &
1685
+ page_pid=$!
1686
+ wait "$page_pid"
1687
+ page_pid=''
1688
+ continue
1689
+ fi
1690
+ wd_log "canary PASS — browser handoff settled; recording deployment proof"
1691
+ if [ -n "$CUTOVER_ID" ]; then
1692
+ cutover_event_required ready "$CUTOVER_ROLE"
1693
+ CUTOVER_ID=''
1694
+ fi
1695
+ if ! guard_cmd record-proven-deployment --repo "$REPO" --state-dir "$STATE_DIR"; then
1696
+ # Proof persistence is an optimization for a later pure restart. This
1697
+ # boot already passed readiness, ownership, and canary; keep it up but
1698
+ # force the next restart back through fresh build/test evidence.
1699
+ wd_log "deployment proof unavailable — the next restart requires fresh build/test evidence" >&2
1700
+ fi
464
1701
  rm -f "$RESTART_MARKER"
465
1702
  else
466
- echo "[watchdog] canary FAIL — rolling back to last known-good"
1703
+ if [ -n "$CUTOVER_ID" ]; then
1704
+ wd_log "canary/ownership FAIL during launch cutover"
1705
+ if [ "$CUTOVER_ROLE" = "target" ]; then
1706
+ cutover_event_required canary target fail "credential/head verification or child/listener identity failed after readiness"
1707
+ elif current_ownership_matches; then
1708
+ # The singleton restart credential normally belongs to the rejected
1709
+ # target repo. Do not relabel that target credential failure as a
1710
+ # previous-host canary failure: stable process/listener ownership is
1711
+ # the explicit recovery proof when no previous-scoped credential is
1712
+ # available.
1713
+ cutover_event_required canary previous skipped \
1714
+ "target-scoped credential is not previous recovery evidence; stable ownership remained proven"
1715
+ else
1716
+ cutover_event_required canary previous fail \
1717
+ "restored child/listener identity failed after readiness"
1718
+ fi
1719
+ if [ "$CUTOVER_ROLE" = "target" ] && [ "$CUTOVER_POLICY" = "restore-previous" ] \
1720
+ && [ -n "$PREVIOUS_START" ] && [ -n "$PREVIOUS_HOME" ] && [ -n "$PREVIOUS_REPO" ] \
1721
+ && [ -n "$PREVIOUS_HARNESS_ROOT" ]; then
1722
+ kill_current_owned_attempt
1723
+ wait "$child" 2>/dev/null || true
1724
+ if select_previous_spec "target became ready but canary failed; restoring previous spec"; then
1725
+ continue
1726
+ fi
1727
+ cutover_event_required awaiting-user "target canary failed and filesystem transition rollback did not complete; previous was not started"
1728
+ park_cutover "target canary failed; previous restore is blocked by filesystem transition rollback"
1729
+ continue
1730
+ fi
1731
+ if [ "$CUTOVER_ROLE" = "previous" ]; then
1732
+ if ! current_ownership_matches; then
1733
+ failures=$((failures + 1))
1734
+ cutover_event_required attempt-failed previous "$current_attempt" \
1735
+ "restored previous ownership changed after readiness"
1736
+ cutover_event_required awaiting-user "restored previous host lost child/listener ownership"
1737
+ wait_cutover_with_live_child
1738
+ continue
1739
+ fi
1740
+ # The previous service is restored and ready; a credential tied to a
1741
+ # different target repo may legitimately fail. Browser acknowledgement
1742
+ # is still required before this recovery becomes terminal.
1743
+ rm -f "$RESTART_MARKER"
1744
+ if complete_browser_handoff; then
1745
+ cutover_event_required ready previous
1746
+ CUTOVER_ID=''
1747
+ else
1748
+ failures=$((failures + 1))
1749
+ cutover_event_required attempt-failed previous "$current_attempt" \
1750
+ "browser handoff failed after restored-previous canary settled"
1751
+ cutover_event_required awaiting-user "restored previous host is ready, but browser handoff was not acknowledged"
1752
+ wait_cutover_with_live_child
1753
+ continue
1754
+ fi
1755
+ else
1756
+ kill_current_owned_attempt
1757
+ wait "$child" 2>/dev/null || true
1758
+ cutover_event_required awaiting-user "target canary failed"
1759
+ printf '%s launch cutover waiting after canary failure\n' "$(date '+%F %T')" > "$GIVE_UP_MARKER"
1760
+ WD_PORT="$PORT" WD_PID="$$" node -e "$(page_script)" &
1761
+ page_pid=$!
1762
+ wait "$page_pid"
1763
+ page_pid=''
1764
+ continue
1765
+ fi
1766
+ else
1767
+ wd_log "canary FAIL — rolling back to last known-good"
467
1768
  sha=$(rollback_sha)
468
1769
  if [ -n "$sha" ]; then rollback_to "$sha" || true; fi
469
1770
  rm -f "$RESTART_MARKER"
470
1771
  failures=0
471
1772
  reset_done=0
472
1773
  continue
1774
+ fi
473
1775
  fi
474
1776
  else
475
1777
  # Unplanned exit (crash, or a stop outside the guard): leave a record the
@@ -489,6 +1791,12 @@ while true; do
489
1791
  fi
490
1792
  fi
491
1793
 
1794
+ # Only a fully ready + canary-settled boot becomes the deployment rollback
1795
+ # target. Stamping before the cutover canary once made a rejected target the
1796
+ # very revision ordinary rollback preferred.
1797
+ stamp_last_good_boot
1798
+ snapshot_composition
1799
+
492
1800
  failures=0
493
1801
  reset_done=0
494
1802
  port_races=0
@@ -501,24 +1809,24 @@ while true; do
501
1809
  if [ ! -f "$PIDFILE" ]; then (set -C; echo $$ > "$PIDFILE") 2>/dev/null || true; fi
502
1810
  pidowner=$(cat "$PIDFILE" 2>/dev/null)
503
1811
  if [ -n "$pidowner" ] && [ "$pidowner" != "$$" ] && kill -0 "$pidowner" 2>/dev/null; then
504
- echo "[watchdog] pidfile now owned by live pid $pidowner — yielding (instance left running for the new owner)"
1812
+ wd_log "pidfile now owned by live pid $pidowner — yielding (instance left running for the new owner)"
505
1813
  yielded=1
506
- exit 0
1814
+ exit 75
507
1815
  fi
508
1816
  fi
509
- sleep 2
1817
+ wd_sleep 2
510
1818
  done
511
1819
  wait "$child"
512
1820
 
513
1821
  # Explicit stop: exit the watchdog without respawn.
514
1822
  if [ -f "$STOP_MARKER" ]; then
515
- echo "[watchdog] stop marker present — exiting"
1823
+ wd_log "stop marker present — exiting"
516
1824
  rm -f "$STOP_MARKER" "$PIDFILE"
517
1825
  exit 0
518
1826
  fi
519
1827
 
520
- sleep 3
521
- if healthy; then
1828
+ wd_sleep 3
1829
+ if transport_up; then
522
1830
  free_port
523
1831
  fi
524
1832
  done