@critical-labs/qa-conductor 0.0.0-stage → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,653 @@
1
+ // Process Provisioner adapter: runs each pane's services, and optionally one
2
+ // database per pane, as local processes on the conductor's machine.
3
+ //
4
+ // The consumer's `command` (and `database.command`) says how to start each
5
+ // one. Every process starts detached, so it leads its own process group, with
6
+ // an env of exactly PATH plus what the pane declares: none of the conductor's
7
+ // own env (its GitHub token, HOME, cloud credentials) reaches a PR's code.
8
+ //
9
+ // Cleanup acts on process groups, never on bare pids, because a wrapper can
10
+ // exit at once while the server it started lives on. Every spawn is recorded
11
+ // in `${stateDir}/pids.json` with the leader's start time (from ps) and the
12
+ // boot it ran in, so a later sweep can tell an orphaned group of ours from an
13
+ // unrelated process that has reused the pid. A leader that exits while its
14
+ // pane is up gives up its pid too, so teardown and the exit hook re-check
15
+ // before signalling one.
16
+ //
17
+ // The env scrubbing and the loopback host are defence in depth only. The
18
+ // BuildConvention's trust gate is the one real boundary between a PR's code
19
+ // and this machine: these processes run as the conductor's user.
20
+ //
21
+ // Every effect (spawn, kill, ps, the boot id, fetch, the free-port lookup, fs,
22
+ // sleep, the clock and the exit hook) is injected with a real default, so the
23
+ // tests start no processes.
24
+
25
+ import { spawn, execFile } from 'node:child_process'
26
+ import { promises as fs } from 'node:fs'
27
+ import net from 'node:net'
28
+ import { StringDecoder } from 'node:string_decoder'
29
+
30
+ const PROBE_MS = 100
31
+ const KILL_WAIT_MS = 2000
32
+ const ATTEMPT_MAX_MS = 2000
33
+ const RETRY_MS = 250
34
+ const PORT_ATTEMPTS = 20
35
+ const MAX_LINE = 8192
36
+ const TAIL_LINES = 40
37
+
38
+ const abortError = () => Object.assign(new Error('aborted'), { name: 'AbortError' })
39
+
40
+ // kill(-pid) signals a whole group. -1 would signal every process we may
41
+ // signal and -0 our own group, so only pids above 1 are ever used.
42
+ const killablePid = pid => Number.isSafeInteger(pid) && pid > 1
43
+
44
+ const startError = (name, cmd, cwd, err) =>
45
+ new Error(`${name}: failed to start ${cmd} in ${cwd ?? '.'}: ${err?.message ?? err}`)
46
+
47
+ function describeExit({ error, code, signal }) {
48
+ if (error) return error.message
49
+ return signal ? `signal ${signal}` : `code ${code}`
50
+ }
51
+
52
+ // The first of `racers(signal)` to settle, or undefined once any of `signals`
53
+ // aborts. `signal` aborts as the race ends and the listeners come off
54
+ // `signals` with it, so a long-lived signal gains nothing per race and no
55
+ // racer that honours its signal outlives the race.
56
+ async function race(signals, racers) {
57
+ const live = signals.filter(Boolean)
58
+ if (live.some(s => s.aborted)) return undefined
59
+ const ctl = new AbortController()
60
+ const stop = () => ctl.abort()
61
+ const stopped = new Promise(resolve => ctl.signal.addEventListener('abort', () => resolve(undefined), { once: true }))
62
+ for (const s of live) s.addEventListener('abort', stop, { once: true })
63
+ try {
64
+ return await Promise.race([...racers(ctl.signal), stopped])
65
+ } finally {
66
+ ctl.abort()
67
+ for (const s of live) s.removeEventListener('abort', stop)
68
+ }
69
+ }
70
+
71
+ // --- real defaults ------------------------------------------------------------
72
+
73
+ // The optional signal clears the timer once a race no longer needs it.
74
+ function defaultSleep(ms, { signal } = {}) {
75
+ return new Promise(resolve => {
76
+ if (signal?.aborted) return resolve()
77
+ const timer = setTimeout(resolve, ms)
78
+ signal?.addEventListener('abort', () => { clearTimeout(timer); resolve() }, { once: true })
79
+ })
80
+ }
81
+
82
+ // A fixed locale and zone: start times are compared as text, so one recorded
83
+ // before the machine's zone changed must still match after.
84
+ function execText(cmd, args) {
85
+ return new Promise((resolve, reject) => {
86
+ execFile(cmd, args, { env: { PATH: process.env.PATH, LC_ALL: 'C', TZ: 'UTC' } }, (err, stdout) => {
87
+ if (err) reject(err)
88
+ else resolve(String(stdout))
89
+ })
90
+ })
91
+ }
92
+
93
+ // The leader's start time, to the second: fixed for the life of a process, and
94
+ // different for any later process that reuses its pid.
95
+ async function defaultPs(pid) {
96
+ try {
97
+ const startedAt = (await execText('ps', ['-o', 'lstart=', '-p', String(pid)])).trim()
98
+ return startedAt ? { startedAt } : null
99
+ } catch (err) {
100
+ if (err.code === 1) return null // ps exits 1 when there is no such process
101
+ throw err
102
+ }
103
+ }
104
+
105
+ async function defaultPgroup(pgid) {
106
+ const members = []
107
+ for (const line of (await execText('ps', ['-A', '-o', 'pid=,pgid='])).split('\n')) {
108
+ const [pid, group] = line.trim().split(/\s+/).map(Number)
109
+ if (group === pgid && Number.isSafeInteger(pid)) members.push(pid)
110
+ }
111
+ return members
112
+ }
113
+
114
+ // Different on every boot, and nothing survives one. macOS has a uuid per
115
+ // boot; other BSDs only the boot time.
116
+ async function defaultBootId() {
117
+ if (process.platform === 'linux') return (await fs.readFile('/proc/sys/kernel/random/boot_id', 'utf8')).trim()
118
+ return (await execText('sysctl', ['-n', process.platform === 'darwin' ? 'kern.bootsessionuuid' : 'kern.boottime'])).trim()
119
+ }
120
+
121
+ function defaultFreePort(host) {
122
+ return new Promise((resolve, reject) => {
123
+ const server = net.createServer()
124
+ server.unref()
125
+ server.once('error', reject)
126
+ server.listen(0, host, () => {
127
+ const { port } = server.address()
128
+ server.close(() => resolve(port))
129
+ })
130
+ })
131
+ }
132
+
133
+ const defaultFsx = {
134
+ mkdir: fs.mkdir,
135
+ lstat: fs.lstat,
136
+ chmod: fs.chmod,
137
+ readFile: fs.readFile,
138
+ writeFile: fs.writeFile,
139
+ rename: fs.rename,
140
+ unlink: fs.unlink,
141
+ }
142
+
143
+ // --- the Provisioner ------------------------------------------------------------
144
+
145
+ export function createProcessProvisioner({
146
+ // ({ name, ref, port, env, paneRef }) => { cmd, args, cwd, env? }
147
+ command,
148
+ // optional: { command({ paneRef, port }), ready({ port, signal }), handle({ paneRef, port }) => { dsn, db } }
149
+ database = null,
150
+ healthPath = '/',
151
+ healthy = status => status < 500,
152
+ healthTimeoutMs = 60000,
153
+ // the pidfile's directory: created 0700, must be a real directory we own
154
+ stateDir,
155
+ // the address in reserved urls, health checks and the free-port lookup
156
+ host = '127.0.0.1',
157
+ graceMs = 5000,
158
+ logLines = 200,
159
+ spawnFn = spawn,
160
+ killFn = (pid, sig) => process.kill(pid, sig),
161
+ freePortFn = defaultFreePort,
162
+ fetchFn = fetch,
163
+ fsx = defaultFsx,
164
+ psFn = defaultPs,
165
+ pgroupFn = defaultPgroup,
166
+ bootIdFn = defaultBootId,
167
+ sleepFn = defaultSleep,
168
+ nowFn = Date.now,
169
+ onExitFn = fn => process.on('exit', fn),
170
+ baseEnv = process.env,
171
+ log = console,
172
+ } = {}) {
173
+ if (typeof command !== 'function') throw new Error('createProcessProvisioner: command must be a function')
174
+ if (typeof stateDir !== 'string' || !stateDir) throw new Error('createProcessProvisioner: stateDir is required')
175
+ if (database && !['command', 'ready', 'handle'].every(k => typeof database[k] === 'function')) {
176
+ throw new Error('createProcessProvisioner: database needs command, ready and handle functions')
177
+ }
178
+
179
+ const pidfile = `${stateDir}/pids.json`
180
+ const urlHost = host.includes(':') ? `[${host}]` : host
181
+ // role -> { procs, logs, ports, stale }. `procs` holds every process started
182
+ // for the pane that isn't yet known to be gone.
183
+ const panes = new Map()
184
+ // Nothing binds a reserved port until launch, so the free-port lookup could
185
+ // offer it again meanwhile; a port stays taken until its pane is torn down.
186
+ const allocated = new Set()
187
+ // service port -> its process, so waitHealthy can watch for an early exit
188
+ const byPort = new Map()
189
+ let exitHooked = false
190
+ let dirReady = null
191
+ let bootIdReady = null
192
+ let pidfileQueue = Promise.resolve()
193
+
194
+ function pane(role) {
195
+ let st = panes.get(role)
196
+ if (!st) panes.set(role, (st = { procs: [], logs: [], ports: new Set(), stale: false }))
197
+ return st
198
+ }
199
+
200
+ // null when it can't be read; the sweep then can't rule out a reboot
201
+ function bootId() {
202
+ bootIdReady ??= Promise.resolve()
203
+ .then(() => bootIdFn())
204
+ .then(id => (typeof id === 'string' && id ? id : null), () => null)
205
+ .then(id => {
206
+ if (!id) log.warn('[qa] no boot id: the sweep cannot tell entries from an earlier boot')
207
+ return id
208
+ })
209
+ return bootIdReady
210
+ }
211
+
212
+ // --- state dir + pidfile ---
213
+
214
+ function stateDirReady() {
215
+ dirReady ??= checkStateDir().catch(err => {
216
+ dirReady = null
217
+ throw err
218
+ })
219
+ return dirReady
220
+ }
221
+
222
+ // The pidfile decides what gets killed, so its directory must be ours alone.
223
+ async function checkStateDir() {
224
+ await fsx.mkdir(stateDir, { recursive: true, mode: 0o700 })
225
+ const st = await fsx.lstat(stateDir)
226
+ if (st.isSymbolicLink() || !st.isDirectory()) throw new Error(`stateDir ${stateDir} must be a directory, not a symlink`)
227
+ const uid = process.getuid?.()
228
+ if (uid !== undefined && st.uid !== uid) throw new Error(`stateDir ${stateDir} must be owned by uid ${uid}, not ${st.uid}`)
229
+ // Whatever another user could have put in it can't be trusted, so no chmod fixes this.
230
+ if (st.mode & 0o022) {
231
+ throw new Error(`stateDir ${stateDir} is writable by group or others (mode ${(st.mode & 0o777).toString(8)}): check its contents, then chmod it 0700`)
232
+ }
233
+ if (st.mode & 0o077) await fsx.chmod(stateDir, 0o700)
234
+ }
235
+
236
+ async function readPidfile() {
237
+ let st
238
+ try {
239
+ st = await fsx.lstat(pidfile)
240
+ } catch {
241
+ return [] // missing: nothing to act on
242
+ }
243
+ const uid = process.getuid?.()
244
+ if (!st.isFile() || (uid !== undefined && st.uid !== uid)) {
245
+ log.warn(`[qa] ${pidfile} is not a regular file owned by uid ${uid}; ignored`)
246
+ return []
247
+ }
248
+ try {
249
+ const list = JSON.parse(String(await fsx.readFile(pidfile, 'utf8')))
250
+ return Array.isArray(list) ? list.filter(e => e && typeof e === 'object') : []
251
+ } catch {
252
+ return [] // corrupt: nothing to act on
253
+ }
254
+ }
255
+
256
+ // Launches, teardowns and the sweep all edit the pidfile. One edit runs at a
257
+ // time and each re-reads the file, so none loses another's entries.
258
+ function editPidfile(edit) {
259
+ const run = pidfileQueue.then(async () => {
260
+ await stateDirReady()
261
+ const tmp = `${pidfile}.${process.pid}.tmp`
262
+ const data = `${JSON.stringify(edit(await readPidfile()), null, 2)}\n`
263
+ // created afresh, so a file or symlink left at the tmp path is replaced, never written through
264
+ await fsx.unlink(tmp).catch(err => {
265
+ if (err?.code !== 'ENOENT') throw err
266
+ })
267
+ await fsx.writeFile(tmp, data, { mode: 0o600, flag: 'wx' })
268
+ await fsx.rename(tmp, pidfile)
269
+ })
270
+ pidfileQueue = run.catch(() => {})
271
+ return run
272
+ }
273
+
274
+ const sameEntry = (a, b) => a.pid === b.pid && a.startedAt === b.startedAt
275
+ const without = done => list => list.filter(e => !done.some(d => sameEntry(e, d)))
276
+ const entryOf = (proc, boot) => ({
277
+ pid: proc.pid, startedAt: proc.startedAt, bootId: boot, paneRole: proc.role, kind: proc.kind, name: proc.name, cmd: proc.cmd, args: proc.args,
278
+ })
279
+
280
+ async function persist(proc) {
281
+ const [cur, boot] = await Promise.all([Promise.resolve().then(() => psFn(proc.pid)).catch(() => null), bootId()])
282
+ proc.startedAt = cur?.startedAt ?? null
283
+ if (proc.gone) return // torn down while ps ran: no group left to record
284
+ if (proc.startedAt === null) {
285
+ log.warn(`[qa] ${proc.role} ${proc.name}: could not read the start time of pid ${proc.pid}; a later sweep will leave its group alone`)
286
+ }
287
+ await editPidfile(list => [...list, entryOf(proc, boot)])
288
+ }
289
+
290
+ // --- logs ---
291
+
292
+ function pushLine(st, line) {
293
+ st.logs.push(line)
294
+ if (st.logs.length > logLines) st.logs.splice(0, st.logs.length - logLines)
295
+ }
296
+
297
+ function resetLogs(st) {
298
+ st.logs = []
299
+ st.stale = false
300
+ }
301
+
302
+ // One output stream's lines into the pane's ring buffer, prefixed [name].
303
+ function capture(st, name, stream) {
304
+ if (!stream) return
305
+ const decoder = new StringDecoder('utf8')
306
+ let partial = ''
307
+ const take = text => {
308
+ const lines = (partial + text).split(/\r?\n/)
309
+ partial = lines.pop()
310
+ // a process that never prints a newline mustn't grow this without bound
311
+ if (partial.length > MAX_LINE) {
312
+ lines.push(partial)
313
+ partial = ''
314
+ }
315
+ for (const line of lines) pushLine(st, `[${name}] ${line}`)
316
+ }
317
+ stream.on('data', chunk => take(decoder.write(chunk)))
318
+ stream.on('end', () => {
319
+ take(decoder.end())
320
+ if (partial) pushLine(st, `[${name}] ${partial}`)
321
+ partial = ''
322
+ })
323
+ stream.on('error', () => {}) // a broken pipe must not crash the conductor
324
+ }
325
+
326
+ function tail(role, lines = TAIL_LINES) {
327
+ const buf = panes.get(role)?.logs ?? []
328
+ return lines > 0 ? buf.slice(-lines).join('\n') : ''
329
+ }
330
+
331
+ const withTail = (err, role) => Object.assign(err, { logTail: tail(role) })
332
+
333
+ // --- spawning ---
334
+
335
+ // Exactly PATH plus the declared layers: nothing else of baseEnv.
336
+ function childEnv(...layers) {
337
+ const env = {}
338
+ for (const [key, value] of Object.entries(Object.assign({ PATH: baseEnv.PATH }, ...layers))) {
339
+ if (value !== undefined) env[key] = value
340
+ }
341
+ return env
342
+ }
343
+
344
+ // Starts one process. It's recorded, and every listener attached, in the
345
+ // same tick as the spawn: a teardown racing this boot always sees it, and
346
+ // no 'error' event goes unheard. Resolves once its pidfile entry is written.
347
+ function start(role, kind, name, spec, env, port) {
348
+ const st = pane(role)
349
+ const { cmd, args = [], cwd } = spec
350
+ if (!exitHooked) {
351
+ exitHooked = true
352
+ onExitFn(killAllNow)
353
+ }
354
+ let child
355
+ try {
356
+ child = spawnFn(cmd, args, { cwd, env, detached: true, stdio: ['ignore', 'pipe', 'pipe'] })
357
+ } catch (err) {
358
+ return Promise.reject(startError(name, cmd, cwd, err))
359
+ }
360
+ const exitCtl = new AbortController()
361
+ const proc = {
362
+ role, kind, name, cmd, args, port, pid: null, startedAt: null,
363
+ exited: false, exit: null, exitSignal: exitCtl.signal, reaped: false, groupEnded: false, gone: false,
364
+ }
365
+ // Node reaps the leader just before 'exit'. From then on its pid can be
366
+ // reissued once the group is empty, so note now whether it already is.
367
+ const reaped = () => {
368
+ if (proc.reaped) return
369
+ proc.reaped = true
370
+ if (killablePid(proc.pid) && groupGone(proc.pid)) proc.groupEnded = true
371
+ }
372
+ // 'error' or 'close' is terminal; 'exit' fires before the output is drained
373
+ proc.done = new Promise(resolve => {
374
+ const finish = exit => {
375
+ if (proc.exited) return
376
+ proc.exited = true
377
+ proc.exit = exit
378
+ exitCtl.abort()
379
+ resolve(exit)
380
+ }
381
+ child.on('error', error => finish({ error }))
382
+ child.on('exit', reaped)
383
+ child.on('close', (code, signal) => {
384
+ reaped() // 'close' always follows 'exit'
385
+ finish({ code, signal })
386
+ })
387
+ })
388
+ st.procs.push(proc)
389
+ if (kind === 'service') byPort.set(port, proc)
390
+ capture(st, name, child.stdout)
391
+ capture(st, name, child.stderr)
392
+ if (!(Number.isSafeInteger(child.pid) && child.pid > 0)) {
393
+ // no pid: the spawn failed, and node says why in 'error'
394
+ return proc.done.then(exit => { throw startError(name, cmd, cwd, exit.error ?? new Error(describeExit(exit))) })
395
+ }
396
+ proc.pid = child.pid
397
+ return persist(proc).then(() => proc)
398
+ }
399
+
400
+ async function allocatePort(st) {
401
+ for (let i = 0; i < PORT_ATTEMPTS; i++) {
402
+ const port = await freePortFn(host)
403
+ if (allocated.has(port)) continue
404
+ allocated.add(port)
405
+ st.ports.add(port)
406
+ return port
407
+ }
408
+ throw new Error(`no free port on ${host} after ${PORT_ATTEMPTS} tries`)
409
+ }
410
+
411
+ // --- stopping ---
412
+
413
+ // 'exit' listeners may only do synchronous work: SIGKILL, with no grace.
414
+ // Signals don't reach here; CLIs call the conductor's shutdown() on those.
415
+ function killAllNow() {
416
+ for (const st of panes.values()) {
417
+ for (const proc of st.procs) {
418
+ // A group that was empty when its leader was reaped may have lost its
419
+ // pid to someone else, and there's no time here for ps: its entry
420
+ // stays in the pidfile for the next sweep. A reaped leader whose group
421
+ // still had members (a wrapper's orphans) is still killed.
422
+ if (!killablePid(proc.pid) || proc.groupEnded) continue
423
+ try {
424
+ killFn(-proc.pid, 'SIGKILL')
425
+ } catch {
426
+ // already gone
427
+ }
428
+ }
429
+ }
430
+ }
431
+
432
+ // Gone once the group can't be signalled: ESRCH means it has no members, and
433
+ // EPERM is what macOS reports for a group left holding only zombies.
434
+ function groupGone(pid) {
435
+ try {
436
+ killFn(-pid, 0)
437
+ return false
438
+ } catch (err) {
439
+ return err?.code === 'ESRCH' || err?.code === 'EPERM'
440
+ }
441
+ }
442
+
443
+ async function waitGone(pid, ms, proc) {
444
+ const deadline = nowFn() + ms
445
+ for (;;) {
446
+ // once our group is known to have ended, its pid may already be someone else's
447
+ if (proc?.groupEnded || groupGone(pid)) return true
448
+ if (nowFn() >= deadline) return false
449
+ await sleepFn(PROBE_MS)
450
+ }
451
+ }
452
+
453
+ // TERM, then KILL once graceMs passes. True once the group is gone.
454
+ async function stopGroup(pid, proc) {
455
+ try {
456
+ killFn(-pid, 'SIGTERM')
457
+ } catch (err) {
458
+ if (err?.code === 'ESRCH') return true
459
+ }
460
+ if (await waitGone(pid, graceMs, proc)) return true
461
+ try {
462
+ killFn(-pid, 'SIGKILL')
463
+ } catch (err) {
464
+ if (err?.code === 'ESRCH') return true
465
+ }
466
+ return waitGone(pid, KILL_WAIT_MS, proc)
467
+ }
468
+
469
+ // True when our group is known to be gone without signalling it; undefined
470
+ // when that can't be checked. Once the leader is reaped, any process holding
471
+ // its pid is someone else's, and a pid isn't reissued while its group has
472
+ // members: our group is gone. With no holder, the group may still hold a
473
+ // wrapper's orphans, and any members are ours.
474
+ async function ended(proc) {
475
+ if (proc.groupEnded) return true
476
+ if (!proc.reaped) return false
477
+ let cur
478
+ try {
479
+ cur = await psFn(proc.pid)
480
+ } catch {
481
+ return undefined
482
+ }
483
+ return cur != null
484
+ }
485
+
486
+ // True once a pidfile entry needs no more sweeping; the entry has a valid
487
+ // pid and a start time.
488
+ async function sweepEntry(entry) {
489
+ const cur = await psFn(entry.pid)
490
+ // A reused pid means our group is long gone: a pid isn't reissued while
491
+ // its process group still has members.
492
+ if (cur) return cur.startedAt === entry.startedAt ? stopGroup(entry.pid) : true
493
+ // The leader is gone, but its group may live on; any members are ours.
494
+ const members = await pgroupFn(entry.pid)
495
+ return members?.length ? stopGroup(entry.pid) : true
496
+ }
497
+
498
+ const ownedHere = entry =>
499
+ [...panes.values()].some(st => st.procs.some(proc => sameEntry(proc, entry)))
500
+
501
+ // --- health ---
502
+
503
+ async function waitOne(name, { url, port }, signal) {
504
+ const proc = byPort.get(port)
505
+ const stops = [signal, proc?.exitSignal]
506
+ const path = typeof healthPath === 'function' ? healthPath(name) : healthPath
507
+ const target = `${url ?? `http://${urlHost}:${port}`}${path}`
508
+ const deadline = nowFn() + healthTimeoutMs
509
+ const check = () => {
510
+ if (signal?.aborted) throw abortError()
511
+ if (proc?.exited) throw withTail(new Error(`${name} exited before it was healthy (${describeExit(proc.exit)})`), proc.role)
512
+ }
513
+ for (;;) {
514
+ check()
515
+ // The fetch's signal aborts however the attempt ends, so a hung fetch
516
+ // never outlives it. Redirects aren't followed: an app often redirects
517
+ // to its public origin, the pane proxy, which answers 503 until the
518
+ // boot is done, so following one would judge the proxy, not the service.
519
+ const status = await race(stops, attempt => [
520
+ (async () => (await fetchFn(target, { signal: attempt, redirect: 'manual' }))?.status)().catch(() => undefined),
521
+ sleepFn(Math.max(0, Math.min(ATTEMPT_MAX_MS, deadline - nowFn())), { signal: attempt }).then(() => undefined),
522
+ ])
523
+ check()
524
+ if (status !== undefined && healthy(status)) return
525
+ if (nowFn() >= deadline) {
526
+ const err = new Error(`${name} on port ${port} not healthy after ${healthTimeoutMs}ms`)
527
+ throw proc ? withTail(err, proc.role) : err
528
+ }
529
+ await race(stops, pause => [sleepFn(RETRY_MS, { signal: pause })])
530
+ }
531
+ }
532
+
533
+ return {
534
+ async provisionDatabase({ paneRef, signal } = {}) {
535
+ const st = pane(paneRef.role)
536
+ resetLogs(st) // a new boot for this pane
537
+ if (!database) return { dsn: null, db: null }
538
+ await stateDirReady()
539
+ const port = await allocatePort(st)
540
+ if (signal?.aborted) throw abortError()
541
+ const spec = database.command({ paneRef, port })
542
+ const proc = await start(paneRef.role, 'database', 'database', spec, childEnv(spec.env), port)
543
+ const readySignal = signal ? AbortSignal.any([signal, proc.exitSignal]) : proc.exitSignal
544
+ try {
545
+ // raced too, so a ready check that ignores its signal can't hang the boot
546
+ await race([signal, proc.exitSignal], () => [(async () => database.ready({ port, signal: readySignal }))()])
547
+ } catch (err) {
548
+ if (!proc.exited && !signal?.aborted) throw err
549
+ }
550
+ // an abort wins: the teardown that follows one may be what stopped it
551
+ if (signal?.aborted) throw abortError()
552
+ if (proc.exited) {
553
+ throw withTail(new Error(`database exited before it was ready (${describeExit(proc.exit)})`), paneRef.role)
554
+ }
555
+ return database.handle({ paneRef, port })
556
+ },
557
+
558
+ async reserveServices({ paneRef, services }) {
559
+ const st = pane(paneRef.role)
560
+ const out = {}
561
+ for (const name of Object.keys(services)) {
562
+ const port = await allocatePort(st)
563
+ out[name] = { url: `http://${urlHost}:${port}`, port }
564
+ }
565
+ return out
566
+ },
567
+
568
+ async launchServices({ paneRef, services, env = {}, reserved, signal } = {}) {
569
+ const st = pane(paneRef.role)
570
+ if (st.stale) resetLogs(st) // the first launch since a teardown starts a new buffer
571
+ await stateDirReady()
572
+ for (const name of Object.keys(services)) {
573
+ const port = reserved?.[name]?.port
574
+ if (!port) throw new Error(`${name}: no reserved port (reserveServices first)`)
575
+ if (signal?.aborted) throw abortError()
576
+ const spec = command({ name, ref: services[name], port, env: env?.[name] ?? {}, paneRef })
577
+ await start(paneRef.role, 'service', name, spec, childEnv(env?.[name], spec.env), port)
578
+ }
579
+ },
580
+
581
+ async waitHealthy({ services, signal } = {}) {
582
+ for (const [name, svc] of Object.entries(services ?? {})) {
583
+ if (svc?.port) await waitOne(name, svc, signal)
584
+ }
585
+ },
586
+
587
+ // One buffer per pane, database and services together, each line
588
+ // prefixed [name]; `stage` doesn't narrow it.
589
+ async logs({ paneRef, lines = TAIL_LINES } = {}) {
590
+ return tail(paneRef?.role, lines)
591
+ },
592
+
593
+ // Orphans a previous run left behind. Entries of this instance's live
594
+ // processes are left alone; only the entries handled here are removed.
595
+ async sweep() {
596
+ await stateDirReady()
597
+ const boot = await bootId()
598
+ const done = []
599
+ for (const entry of await readPidfile()) {
600
+ if (ownedHere(entry)) continue
601
+ // Nothing outlives a reboot, so an entry from an earlier one names only
602
+ // dead processes, whatever holds its pid (or pgid) now.
603
+ if (boot && entry.bootId !== boot) {
604
+ done.push(entry)
605
+ continue
606
+ }
607
+ // Nothing to check it against: kept, until a reboot clears it.
608
+ if (!killablePid(entry.pid) || typeof entry.startedAt !== 'string') continue
609
+ try {
610
+ if (await sweepEntry(entry)) done.push(entry)
611
+ else log.warn(`[qa] sweep: process group ${entry.pid} (${entry.name}) survived SIGKILL; kept in ${pidfile}`)
612
+ } catch (err) {
613
+ log.warn(`[qa] sweep: skipped pid ${entry.pid}: ${err.message}`)
614
+ }
615
+ }
616
+ if (done.length) await editPidfile(without(done))
617
+ },
618
+
619
+ async teardown({ paneRef }) {
620
+ const st = panes.get(paneRef.role)
621
+ if (!st) return
622
+ st.stale = true
623
+ // services before the database they depend on
624
+ const order = [...st.procs.filter(p => p.kind !== 'database'), ...st.procs.filter(p => p.kind === 'database')]
625
+ const gone = []
626
+ for (const proc of order) {
627
+ if (killablePid(proc.pid)) {
628
+ const known = await ended(proc)
629
+ if (known === undefined) {
630
+ log.warn(`[qa] ${paneRef.role} ${proc.name}: could not check whether pid ${proc.pid} is still ours; not signalled, kept in ${pidfile}`)
631
+ continue
632
+ }
633
+ if (!known && !(await stopGroup(proc.pid, proc))) {
634
+ log.warn(`[qa] ${paneRef.role} ${proc.name}: process group ${proc.pid} survived SIGKILL; kept in ${pidfile} for the next sweep`)
635
+ continue
636
+ }
637
+ }
638
+ proc.gone = true
639
+ gone.push(proc)
640
+ }
641
+ st.procs = st.procs.filter(proc => !proc.gone)
642
+ for (const proc of gone) if (byPort.get(proc.port) === proc) byPort.delete(proc.port)
643
+ const held = new Set(st.procs.map(proc => proc.port))
644
+ for (const port of st.ports) {
645
+ if (held.has(port)) continue
646
+ allocated.delete(port)
647
+ st.ports.delete(port)
648
+ }
649
+ const recorded = gone.filter(proc => proc.pid !== null)
650
+ if (recorded.length) await editPidfile(without(recorded))
651
+ },
652
+ }
653
+ }