@critical-labs/qa-conductor 0.0.0-stage → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +445 -2
- package/bin/qa-conductor-expose.mjs +113 -0
- package/lib/adapters/build-worktree.mjs +532 -0
- package/lib/adapters/exposure-tailscale.mjs +136 -0
- package/lib/adapters/provisioner-docker.mjs +116 -0
- package/lib/adapters/provisioner-process.mjs +653 -0
- package/lib/config.mjs +245 -0
- package/lib/docker.mjs +233 -0
- package/lib/exec.mjs +23 -0
- package/lib/exposure.mjs +193 -0
- package/lib/github.mjs +218 -0
- package/lib/identity.mjs +51 -0
- package/lib/net.mjs +56 -0
- package/lib/proxy.mjs +327 -0
- package/lib/request-guard.mjs +95 -0
- package/lib/server.mjs +721 -0
- package/lib/session.mjs +250 -0
- package/lib/verdict.mjs +33 -0
- package/package.json +47 -4
- package/public/bridge.js +324 -0
- package/public/harness.js +542 -0
- package/public/index.html +284 -0
|
@@ -0,0 +1,653 @@
|
|
|
1
|
+
// Process Provisioner adapter: runs each pane's services, and optionally one
|
|
2
|
+
// database per pane, as local processes on the conductor's machine.
|
|
3
|
+
//
|
|
4
|
+
// The consumer's `command` (and `database.command`) says how to start each
|
|
5
|
+
// one. Every process starts detached, so it leads its own process group, with
|
|
6
|
+
// an env of exactly PATH plus what the pane declares: none of the conductor's
|
|
7
|
+
// own env (its GitHub token, HOME, cloud credentials) reaches a PR's code.
|
|
8
|
+
//
|
|
9
|
+
// Cleanup acts on process groups, never on bare pids, because a wrapper can
|
|
10
|
+
// exit at once while the server it started lives on. Every spawn is recorded
|
|
11
|
+
// in `${stateDir}/pids.json` with the leader's start time (from ps) and the
|
|
12
|
+
// boot it ran in, so a later sweep can tell an orphaned group of ours from an
|
|
13
|
+
// unrelated process that has reused the pid. A leader that exits while its
|
|
14
|
+
// pane is up gives up its pid too, so teardown and the exit hook re-check
|
|
15
|
+
// before signalling one.
|
|
16
|
+
//
|
|
17
|
+
// The env scrubbing and the loopback host are defence in depth only. The
|
|
18
|
+
// BuildConvention's trust gate is the one real boundary between a PR's code
|
|
19
|
+
// and this machine: these processes run as the conductor's user.
|
|
20
|
+
//
|
|
21
|
+
// Every effect (spawn, kill, ps, the boot id, fetch, the free-port lookup, fs,
|
|
22
|
+
// sleep, the clock and the exit hook) is injected with a real default, so the
|
|
23
|
+
// tests start no processes.
|
|
24
|
+
|
|
25
|
+
import { spawn, execFile } from 'node:child_process'
|
|
26
|
+
import { promises as fs } from 'node:fs'
|
|
27
|
+
import net from 'node:net'
|
|
28
|
+
import { StringDecoder } from 'node:string_decoder'
|
|
29
|
+
|
|
30
|
+
const PROBE_MS = 100
|
|
31
|
+
const KILL_WAIT_MS = 2000
|
|
32
|
+
const ATTEMPT_MAX_MS = 2000
|
|
33
|
+
const RETRY_MS = 250
|
|
34
|
+
const PORT_ATTEMPTS = 20
|
|
35
|
+
const MAX_LINE = 8192
|
|
36
|
+
const TAIL_LINES = 40
|
|
37
|
+
|
|
38
|
+
const abortError = () => Object.assign(new Error('aborted'), { name: 'AbortError' })
|
|
39
|
+
|
|
40
|
+
// kill(-pid) signals a whole group. -1 would signal every process we may
|
|
41
|
+
// signal and -0 our own group, so only pids above 1 are ever used.
|
|
42
|
+
const killablePid = pid => Number.isSafeInteger(pid) && pid > 1
|
|
43
|
+
|
|
44
|
+
const startError = (name, cmd, cwd, err) =>
|
|
45
|
+
new Error(`${name}: failed to start ${cmd} in ${cwd ?? '.'}: ${err?.message ?? err}`)
|
|
46
|
+
|
|
47
|
+
function describeExit({ error, code, signal }) {
|
|
48
|
+
if (error) return error.message
|
|
49
|
+
return signal ? `signal ${signal}` : `code ${code}`
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
// The first of `racers(signal)` to settle, or undefined once any of `signals`
|
|
53
|
+
// aborts. `signal` aborts as the race ends and the listeners come off
|
|
54
|
+
// `signals` with it, so a long-lived signal gains nothing per race and no
|
|
55
|
+
// racer that honours its signal outlives the race.
|
|
56
|
+
async function race(signals, racers) {
|
|
57
|
+
const live = signals.filter(Boolean)
|
|
58
|
+
if (live.some(s => s.aborted)) return undefined
|
|
59
|
+
const ctl = new AbortController()
|
|
60
|
+
const stop = () => ctl.abort()
|
|
61
|
+
const stopped = new Promise(resolve => ctl.signal.addEventListener('abort', () => resolve(undefined), { once: true }))
|
|
62
|
+
for (const s of live) s.addEventListener('abort', stop, { once: true })
|
|
63
|
+
try {
|
|
64
|
+
return await Promise.race([...racers(ctl.signal), stopped])
|
|
65
|
+
} finally {
|
|
66
|
+
ctl.abort()
|
|
67
|
+
for (const s of live) s.removeEventListener('abort', stop)
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
// --- real defaults ------------------------------------------------------------
|
|
72
|
+
|
|
73
|
+
// The optional signal clears the timer once a race no longer needs it.
|
|
74
|
+
function defaultSleep(ms, { signal } = {}) {
|
|
75
|
+
return new Promise(resolve => {
|
|
76
|
+
if (signal?.aborted) return resolve()
|
|
77
|
+
const timer = setTimeout(resolve, ms)
|
|
78
|
+
signal?.addEventListener('abort', () => { clearTimeout(timer); resolve() }, { once: true })
|
|
79
|
+
})
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// A fixed locale and zone: start times are compared as text, so one recorded
|
|
83
|
+
// before the machine's zone changed must still match after.
|
|
84
|
+
function execText(cmd, args) {
|
|
85
|
+
return new Promise((resolve, reject) => {
|
|
86
|
+
execFile(cmd, args, { env: { PATH: process.env.PATH, LC_ALL: 'C', TZ: 'UTC' } }, (err, stdout) => {
|
|
87
|
+
if (err) reject(err)
|
|
88
|
+
else resolve(String(stdout))
|
|
89
|
+
})
|
|
90
|
+
})
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// The leader's start time, to the second: fixed for the life of a process, and
|
|
94
|
+
// different for any later process that reuses its pid.
|
|
95
|
+
async function defaultPs(pid) {
|
|
96
|
+
try {
|
|
97
|
+
const startedAt = (await execText('ps', ['-o', 'lstart=', '-p', String(pid)])).trim()
|
|
98
|
+
return startedAt ? { startedAt } : null
|
|
99
|
+
} catch (err) {
|
|
100
|
+
if (err.code === 1) return null // ps exits 1 when there is no such process
|
|
101
|
+
throw err
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
async function defaultPgroup(pgid) {
|
|
106
|
+
const members = []
|
|
107
|
+
for (const line of (await execText('ps', ['-A', '-o', 'pid=,pgid='])).split('\n')) {
|
|
108
|
+
const [pid, group] = line.trim().split(/\s+/).map(Number)
|
|
109
|
+
if (group === pgid && Number.isSafeInteger(pid)) members.push(pid)
|
|
110
|
+
}
|
|
111
|
+
return members
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
// Different on every boot, and nothing survives one. macOS has a uuid per
|
|
115
|
+
// boot; other BSDs only the boot time.
|
|
116
|
+
async function defaultBootId() {
|
|
117
|
+
if (process.platform === 'linux') return (await fs.readFile('/proc/sys/kernel/random/boot_id', 'utf8')).trim()
|
|
118
|
+
return (await execText('sysctl', ['-n', process.platform === 'darwin' ? 'kern.bootsessionuuid' : 'kern.boottime'])).trim()
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function defaultFreePort(host) {
|
|
122
|
+
return new Promise((resolve, reject) => {
|
|
123
|
+
const server = net.createServer()
|
|
124
|
+
server.unref()
|
|
125
|
+
server.once('error', reject)
|
|
126
|
+
server.listen(0, host, () => {
|
|
127
|
+
const { port } = server.address()
|
|
128
|
+
server.close(() => resolve(port))
|
|
129
|
+
})
|
|
130
|
+
})
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
const defaultFsx = {
|
|
134
|
+
mkdir: fs.mkdir,
|
|
135
|
+
lstat: fs.lstat,
|
|
136
|
+
chmod: fs.chmod,
|
|
137
|
+
readFile: fs.readFile,
|
|
138
|
+
writeFile: fs.writeFile,
|
|
139
|
+
rename: fs.rename,
|
|
140
|
+
unlink: fs.unlink,
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
// --- the Provisioner ------------------------------------------------------------
|
|
144
|
+
|
|
145
|
+
export function createProcessProvisioner({
|
|
146
|
+
// ({ name, ref, port, env, paneRef }) => { cmd, args, cwd, env? }
|
|
147
|
+
command,
|
|
148
|
+
// optional: { command({ paneRef, port }), ready({ port, signal }), handle({ paneRef, port }) => { dsn, db } }
|
|
149
|
+
database = null,
|
|
150
|
+
healthPath = '/',
|
|
151
|
+
healthy = status => status < 500,
|
|
152
|
+
healthTimeoutMs = 60000,
|
|
153
|
+
// the pidfile's directory: created 0700, must be a real directory we own
|
|
154
|
+
stateDir,
|
|
155
|
+
// the address in reserved urls, health checks and the free-port lookup
|
|
156
|
+
host = '127.0.0.1',
|
|
157
|
+
graceMs = 5000,
|
|
158
|
+
logLines = 200,
|
|
159
|
+
spawnFn = spawn,
|
|
160
|
+
killFn = (pid, sig) => process.kill(pid, sig),
|
|
161
|
+
freePortFn = defaultFreePort,
|
|
162
|
+
fetchFn = fetch,
|
|
163
|
+
fsx = defaultFsx,
|
|
164
|
+
psFn = defaultPs,
|
|
165
|
+
pgroupFn = defaultPgroup,
|
|
166
|
+
bootIdFn = defaultBootId,
|
|
167
|
+
sleepFn = defaultSleep,
|
|
168
|
+
nowFn = Date.now,
|
|
169
|
+
onExitFn = fn => process.on('exit', fn),
|
|
170
|
+
baseEnv = process.env,
|
|
171
|
+
log = console,
|
|
172
|
+
} = {}) {
|
|
173
|
+
if (typeof command !== 'function') throw new Error('createProcessProvisioner: command must be a function')
|
|
174
|
+
if (typeof stateDir !== 'string' || !stateDir) throw new Error('createProcessProvisioner: stateDir is required')
|
|
175
|
+
if (database && !['command', 'ready', 'handle'].every(k => typeof database[k] === 'function')) {
|
|
176
|
+
throw new Error('createProcessProvisioner: database needs command, ready and handle functions')
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
const pidfile = `${stateDir}/pids.json`
|
|
180
|
+
const urlHost = host.includes(':') ? `[${host}]` : host
|
|
181
|
+
// role -> { procs, logs, ports, stale }. `procs` holds every process started
|
|
182
|
+
// for the pane that isn't yet known to be gone.
|
|
183
|
+
const panes = new Map()
|
|
184
|
+
// Nothing binds a reserved port until launch, so the free-port lookup could
|
|
185
|
+
// offer it again meanwhile; a port stays taken until its pane is torn down.
|
|
186
|
+
const allocated = new Set()
|
|
187
|
+
// service port -> its process, so waitHealthy can watch for an early exit
|
|
188
|
+
const byPort = new Map()
|
|
189
|
+
let exitHooked = false
|
|
190
|
+
let dirReady = null
|
|
191
|
+
let bootIdReady = null
|
|
192
|
+
let pidfileQueue = Promise.resolve()
|
|
193
|
+
|
|
194
|
+
function pane(role) {
|
|
195
|
+
let st = panes.get(role)
|
|
196
|
+
if (!st) panes.set(role, (st = { procs: [], logs: [], ports: new Set(), stale: false }))
|
|
197
|
+
return st
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
// null when it can't be read; the sweep then can't rule out a reboot
|
|
201
|
+
function bootId() {
|
|
202
|
+
bootIdReady ??= Promise.resolve()
|
|
203
|
+
.then(() => bootIdFn())
|
|
204
|
+
.then(id => (typeof id === 'string' && id ? id : null), () => null)
|
|
205
|
+
.then(id => {
|
|
206
|
+
if (!id) log.warn('[qa] no boot id: the sweep cannot tell entries from an earlier boot')
|
|
207
|
+
return id
|
|
208
|
+
})
|
|
209
|
+
return bootIdReady
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
// --- state dir + pidfile ---
|
|
213
|
+
|
|
214
|
+
function stateDirReady() {
|
|
215
|
+
dirReady ??= checkStateDir().catch(err => {
|
|
216
|
+
dirReady = null
|
|
217
|
+
throw err
|
|
218
|
+
})
|
|
219
|
+
return dirReady
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
// The pidfile decides what gets killed, so its directory must be ours alone.
|
|
223
|
+
async function checkStateDir() {
|
|
224
|
+
await fsx.mkdir(stateDir, { recursive: true, mode: 0o700 })
|
|
225
|
+
const st = await fsx.lstat(stateDir)
|
|
226
|
+
if (st.isSymbolicLink() || !st.isDirectory()) throw new Error(`stateDir ${stateDir} must be a directory, not a symlink`)
|
|
227
|
+
const uid = process.getuid?.()
|
|
228
|
+
if (uid !== undefined && st.uid !== uid) throw new Error(`stateDir ${stateDir} must be owned by uid ${uid}, not ${st.uid}`)
|
|
229
|
+
// Whatever another user could have put in it can't be trusted, so no chmod fixes this.
|
|
230
|
+
if (st.mode & 0o022) {
|
|
231
|
+
throw new Error(`stateDir ${stateDir} is writable by group or others (mode ${(st.mode & 0o777).toString(8)}): check its contents, then chmod it 0700`)
|
|
232
|
+
}
|
|
233
|
+
if (st.mode & 0o077) await fsx.chmod(stateDir, 0o700)
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
async function readPidfile() {
|
|
237
|
+
let st
|
|
238
|
+
try {
|
|
239
|
+
st = await fsx.lstat(pidfile)
|
|
240
|
+
} catch {
|
|
241
|
+
return [] // missing: nothing to act on
|
|
242
|
+
}
|
|
243
|
+
const uid = process.getuid?.()
|
|
244
|
+
if (!st.isFile() || (uid !== undefined && st.uid !== uid)) {
|
|
245
|
+
log.warn(`[qa] ${pidfile} is not a regular file owned by uid ${uid}; ignored`)
|
|
246
|
+
return []
|
|
247
|
+
}
|
|
248
|
+
try {
|
|
249
|
+
const list = JSON.parse(String(await fsx.readFile(pidfile, 'utf8')))
|
|
250
|
+
return Array.isArray(list) ? list.filter(e => e && typeof e === 'object') : []
|
|
251
|
+
} catch {
|
|
252
|
+
return [] // corrupt: nothing to act on
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
// Launches, teardowns and the sweep all edit the pidfile. One edit runs at a
|
|
257
|
+
// time and each re-reads the file, so none loses another's entries.
|
|
258
|
+
function editPidfile(edit) {
|
|
259
|
+
const run = pidfileQueue.then(async () => {
|
|
260
|
+
await stateDirReady()
|
|
261
|
+
const tmp = `${pidfile}.${process.pid}.tmp`
|
|
262
|
+
const data = `${JSON.stringify(edit(await readPidfile()), null, 2)}\n`
|
|
263
|
+
// created afresh, so a file or symlink left at the tmp path is replaced, never written through
|
|
264
|
+
await fsx.unlink(tmp).catch(err => {
|
|
265
|
+
if (err?.code !== 'ENOENT') throw err
|
|
266
|
+
})
|
|
267
|
+
await fsx.writeFile(tmp, data, { mode: 0o600, flag: 'wx' })
|
|
268
|
+
await fsx.rename(tmp, pidfile)
|
|
269
|
+
})
|
|
270
|
+
pidfileQueue = run.catch(() => {})
|
|
271
|
+
return run
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
const sameEntry = (a, b) => a.pid === b.pid && a.startedAt === b.startedAt
|
|
275
|
+
const without = done => list => list.filter(e => !done.some(d => sameEntry(e, d)))
|
|
276
|
+
const entryOf = (proc, boot) => ({
|
|
277
|
+
pid: proc.pid, startedAt: proc.startedAt, bootId: boot, paneRole: proc.role, kind: proc.kind, name: proc.name, cmd: proc.cmd, args: proc.args,
|
|
278
|
+
})
|
|
279
|
+
|
|
280
|
+
async function persist(proc) {
|
|
281
|
+
const [cur, boot] = await Promise.all([Promise.resolve().then(() => psFn(proc.pid)).catch(() => null), bootId()])
|
|
282
|
+
proc.startedAt = cur?.startedAt ?? null
|
|
283
|
+
if (proc.gone) return // torn down while ps ran: no group left to record
|
|
284
|
+
if (proc.startedAt === null) {
|
|
285
|
+
log.warn(`[qa] ${proc.role} ${proc.name}: could not read the start time of pid ${proc.pid}; a later sweep will leave its group alone`)
|
|
286
|
+
}
|
|
287
|
+
await editPidfile(list => [...list, entryOf(proc, boot)])
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
// --- logs ---
|
|
291
|
+
|
|
292
|
+
function pushLine(st, line) {
|
|
293
|
+
st.logs.push(line)
|
|
294
|
+
if (st.logs.length > logLines) st.logs.splice(0, st.logs.length - logLines)
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
function resetLogs(st) {
|
|
298
|
+
st.logs = []
|
|
299
|
+
st.stale = false
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
// One output stream's lines into the pane's ring buffer, prefixed [name].
|
|
303
|
+
function capture(st, name, stream) {
|
|
304
|
+
if (!stream) return
|
|
305
|
+
const decoder = new StringDecoder('utf8')
|
|
306
|
+
let partial = ''
|
|
307
|
+
const take = text => {
|
|
308
|
+
const lines = (partial + text).split(/\r?\n/)
|
|
309
|
+
partial = lines.pop()
|
|
310
|
+
// a process that never prints a newline mustn't grow this without bound
|
|
311
|
+
if (partial.length > MAX_LINE) {
|
|
312
|
+
lines.push(partial)
|
|
313
|
+
partial = ''
|
|
314
|
+
}
|
|
315
|
+
for (const line of lines) pushLine(st, `[${name}] ${line}`)
|
|
316
|
+
}
|
|
317
|
+
stream.on('data', chunk => take(decoder.write(chunk)))
|
|
318
|
+
stream.on('end', () => {
|
|
319
|
+
take(decoder.end())
|
|
320
|
+
if (partial) pushLine(st, `[${name}] ${partial}`)
|
|
321
|
+
partial = ''
|
|
322
|
+
})
|
|
323
|
+
stream.on('error', () => {}) // a broken pipe must not crash the conductor
|
|
324
|
+
}
|
|
325
|
+
|
|
326
|
+
function tail(role, lines = TAIL_LINES) {
|
|
327
|
+
const buf = panes.get(role)?.logs ?? []
|
|
328
|
+
return lines > 0 ? buf.slice(-lines).join('\n') : ''
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
const withTail = (err, role) => Object.assign(err, { logTail: tail(role) })
|
|
332
|
+
|
|
333
|
+
// --- spawning ---
|
|
334
|
+
|
|
335
|
+
// Exactly PATH plus the declared layers: nothing else of baseEnv.
|
|
336
|
+
function childEnv(...layers) {
|
|
337
|
+
const env = {}
|
|
338
|
+
for (const [key, value] of Object.entries(Object.assign({ PATH: baseEnv.PATH }, ...layers))) {
|
|
339
|
+
if (value !== undefined) env[key] = value
|
|
340
|
+
}
|
|
341
|
+
return env
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
// Starts one process. It's recorded, and every listener attached, in the
|
|
345
|
+
// same tick as the spawn: a teardown racing this boot always sees it, and
|
|
346
|
+
// no 'error' event goes unheard. Resolves once its pidfile entry is written.
|
|
347
|
+
function start(role, kind, name, spec, env, port) {
|
|
348
|
+
const st = pane(role)
|
|
349
|
+
const { cmd, args = [], cwd } = spec
|
|
350
|
+
if (!exitHooked) {
|
|
351
|
+
exitHooked = true
|
|
352
|
+
onExitFn(killAllNow)
|
|
353
|
+
}
|
|
354
|
+
let child
|
|
355
|
+
try {
|
|
356
|
+
child = spawnFn(cmd, args, { cwd, env, detached: true, stdio: ['ignore', 'pipe', 'pipe'] })
|
|
357
|
+
} catch (err) {
|
|
358
|
+
return Promise.reject(startError(name, cmd, cwd, err))
|
|
359
|
+
}
|
|
360
|
+
const exitCtl = new AbortController()
|
|
361
|
+
const proc = {
|
|
362
|
+
role, kind, name, cmd, args, port, pid: null, startedAt: null,
|
|
363
|
+
exited: false, exit: null, exitSignal: exitCtl.signal, reaped: false, groupEnded: false, gone: false,
|
|
364
|
+
}
|
|
365
|
+
// Node reaps the leader just before 'exit'. From then on its pid can be
|
|
366
|
+
// reissued once the group is empty, so note now whether it already is.
|
|
367
|
+
const reaped = () => {
|
|
368
|
+
if (proc.reaped) return
|
|
369
|
+
proc.reaped = true
|
|
370
|
+
if (killablePid(proc.pid) && groupGone(proc.pid)) proc.groupEnded = true
|
|
371
|
+
}
|
|
372
|
+
// 'error' or 'close' is terminal; 'exit' fires before the output is drained
|
|
373
|
+
proc.done = new Promise(resolve => {
|
|
374
|
+
const finish = exit => {
|
|
375
|
+
if (proc.exited) return
|
|
376
|
+
proc.exited = true
|
|
377
|
+
proc.exit = exit
|
|
378
|
+
exitCtl.abort()
|
|
379
|
+
resolve(exit)
|
|
380
|
+
}
|
|
381
|
+
child.on('error', error => finish({ error }))
|
|
382
|
+
child.on('exit', reaped)
|
|
383
|
+
child.on('close', (code, signal) => {
|
|
384
|
+
reaped() // 'close' always follows 'exit'
|
|
385
|
+
finish({ code, signal })
|
|
386
|
+
})
|
|
387
|
+
})
|
|
388
|
+
st.procs.push(proc)
|
|
389
|
+
if (kind === 'service') byPort.set(port, proc)
|
|
390
|
+
capture(st, name, child.stdout)
|
|
391
|
+
capture(st, name, child.stderr)
|
|
392
|
+
if (!(Number.isSafeInteger(child.pid) && child.pid > 0)) {
|
|
393
|
+
// no pid: the spawn failed, and node says why in 'error'
|
|
394
|
+
return proc.done.then(exit => { throw startError(name, cmd, cwd, exit.error ?? new Error(describeExit(exit))) })
|
|
395
|
+
}
|
|
396
|
+
proc.pid = child.pid
|
|
397
|
+
return persist(proc).then(() => proc)
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
async function allocatePort(st) {
|
|
401
|
+
for (let i = 0; i < PORT_ATTEMPTS; i++) {
|
|
402
|
+
const port = await freePortFn(host)
|
|
403
|
+
if (allocated.has(port)) continue
|
|
404
|
+
allocated.add(port)
|
|
405
|
+
st.ports.add(port)
|
|
406
|
+
return port
|
|
407
|
+
}
|
|
408
|
+
throw new Error(`no free port on ${host} after ${PORT_ATTEMPTS} tries`)
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
// --- stopping ---
|
|
412
|
+
|
|
413
|
+
// 'exit' listeners may only do synchronous work: SIGKILL, with no grace.
|
|
414
|
+
// Signals don't reach here; CLIs call the conductor's shutdown() on those.
|
|
415
|
+
function killAllNow() {
|
|
416
|
+
for (const st of panes.values()) {
|
|
417
|
+
for (const proc of st.procs) {
|
|
418
|
+
// A group that was empty when its leader was reaped may have lost its
|
|
419
|
+
// pid to someone else, and there's no time here for ps: its entry
|
|
420
|
+
// stays in the pidfile for the next sweep. A reaped leader whose group
|
|
421
|
+
// still had members (a wrapper's orphans) is still killed.
|
|
422
|
+
if (!killablePid(proc.pid) || proc.groupEnded) continue
|
|
423
|
+
try {
|
|
424
|
+
killFn(-proc.pid, 'SIGKILL')
|
|
425
|
+
} catch {
|
|
426
|
+
// already gone
|
|
427
|
+
}
|
|
428
|
+
}
|
|
429
|
+
}
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
// Gone once the group can't be signalled: ESRCH means it has no members, and
|
|
433
|
+
// EPERM is what macOS reports for a group left holding only zombies.
|
|
434
|
+
function groupGone(pid) {
|
|
435
|
+
try {
|
|
436
|
+
killFn(-pid, 0)
|
|
437
|
+
return false
|
|
438
|
+
} catch (err) {
|
|
439
|
+
return err?.code === 'ESRCH' || err?.code === 'EPERM'
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
|
|
443
|
+
async function waitGone(pid, ms, proc) {
|
|
444
|
+
const deadline = nowFn() + ms
|
|
445
|
+
for (;;) {
|
|
446
|
+
// once our group is known to have ended, its pid may already be someone else's
|
|
447
|
+
if (proc?.groupEnded || groupGone(pid)) return true
|
|
448
|
+
if (nowFn() >= deadline) return false
|
|
449
|
+
await sleepFn(PROBE_MS)
|
|
450
|
+
}
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
// TERM, then KILL once graceMs passes. True once the group is gone.
|
|
454
|
+
async function stopGroup(pid, proc) {
|
|
455
|
+
try {
|
|
456
|
+
killFn(-pid, 'SIGTERM')
|
|
457
|
+
} catch (err) {
|
|
458
|
+
if (err?.code === 'ESRCH') return true
|
|
459
|
+
}
|
|
460
|
+
if (await waitGone(pid, graceMs, proc)) return true
|
|
461
|
+
try {
|
|
462
|
+
killFn(-pid, 'SIGKILL')
|
|
463
|
+
} catch (err) {
|
|
464
|
+
if (err?.code === 'ESRCH') return true
|
|
465
|
+
}
|
|
466
|
+
return waitGone(pid, KILL_WAIT_MS, proc)
|
|
467
|
+
}
|
|
468
|
+
|
|
469
|
+
// True when our group is known to be gone without signalling it; undefined
|
|
470
|
+
// when that can't be checked. Once the leader is reaped, any process holding
|
|
471
|
+
// its pid is someone else's, and a pid isn't reissued while its group has
|
|
472
|
+
// members: our group is gone. With no holder, the group may still hold a
|
|
473
|
+
// wrapper's orphans, and any members are ours.
|
|
474
|
+
async function ended(proc) {
|
|
475
|
+
if (proc.groupEnded) return true
|
|
476
|
+
if (!proc.reaped) return false
|
|
477
|
+
let cur
|
|
478
|
+
try {
|
|
479
|
+
cur = await psFn(proc.pid)
|
|
480
|
+
} catch {
|
|
481
|
+
return undefined
|
|
482
|
+
}
|
|
483
|
+
return cur != null
|
|
484
|
+
}
|
|
485
|
+
|
|
486
|
+
// True once a pidfile entry needs no more sweeping; the entry has a valid
|
|
487
|
+
// pid and a start time.
|
|
488
|
+
async function sweepEntry(entry) {
|
|
489
|
+
const cur = await psFn(entry.pid)
|
|
490
|
+
// A reused pid means our group is long gone: a pid isn't reissued while
|
|
491
|
+
// its process group still has members.
|
|
492
|
+
if (cur) return cur.startedAt === entry.startedAt ? stopGroup(entry.pid) : true
|
|
493
|
+
// The leader is gone, but its group may live on; any members are ours.
|
|
494
|
+
const members = await pgroupFn(entry.pid)
|
|
495
|
+
return members?.length ? stopGroup(entry.pid) : true
|
|
496
|
+
}
|
|
497
|
+
|
|
498
|
+
const ownedHere = entry =>
|
|
499
|
+
[...panes.values()].some(st => st.procs.some(proc => sameEntry(proc, entry)))
|
|
500
|
+
|
|
501
|
+
// --- health ---
|
|
502
|
+
|
|
503
|
+
async function waitOne(name, { url, port }, signal) {
|
|
504
|
+
const proc = byPort.get(port)
|
|
505
|
+
const stops = [signal, proc?.exitSignal]
|
|
506
|
+
const path = typeof healthPath === 'function' ? healthPath(name) : healthPath
|
|
507
|
+
const target = `${url ?? `http://${urlHost}:${port}`}${path}`
|
|
508
|
+
const deadline = nowFn() + healthTimeoutMs
|
|
509
|
+
const check = () => {
|
|
510
|
+
if (signal?.aborted) throw abortError()
|
|
511
|
+
if (proc?.exited) throw withTail(new Error(`${name} exited before it was healthy (${describeExit(proc.exit)})`), proc.role)
|
|
512
|
+
}
|
|
513
|
+
for (;;) {
|
|
514
|
+
check()
|
|
515
|
+
// The fetch's signal aborts however the attempt ends, so a hung fetch
|
|
516
|
+
// never outlives it. Redirects aren't followed: an app often redirects
|
|
517
|
+
// to its public origin, the pane proxy, which answers 503 until the
|
|
518
|
+
// boot is done, so following one would judge the proxy, not the service.
|
|
519
|
+
const status = await race(stops, attempt => [
|
|
520
|
+
(async () => (await fetchFn(target, { signal: attempt, redirect: 'manual' }))?.status)().catch(() => undefined),
|
|
521
|
+
sleepFn(Math.max(0, Math.min(ATTEMPT_MAX_MS, deadline - nowFn())), { signal: attempt }).then(() => undefined),
|
|
522
|
+
])
|
|
523
|
+
check()
|
|
524
|
+
if (status !== undefined && healthy(status)) return
|
|
525
|
+
if (nowFn() >= deadline) {
|
|
526
|
+
const err = new Error(`${name} on port ${port} not healthy after ${healthTimeoutMs}ms`)
|
|
527
|
+
throw proc ? withTail(err, proc.role) : err
|
|
528
|
+
}
|
|
529
|
+
await race(stops, pause => [sleepFn(RETRY_MS, { signal: pause })])
|
|
530
|
+
}
|
|
531
|
+
}
|
|
532
|
+
|
|
533
|
+
return {
|
|
534
|
+
async provisionDatabase({ paneRef, signal } = {}) {
|
|
535
|
+
const st = pane(paneRef.role)
|
|
536
|
+
resetLogs(st) // a new boot for this pane
|
|
537
|
+
if (!database) return { dsn: null, db: null }
|
|
538
|
+
await stateDirReady()
|
|
539
|
+
const port = await allocatePort(st)
|
|
540
|
+
if (signal?.aborted) throw abortError()
|
|
541
|
+
const spec = database.command({ paneRef, port })
|
|
542
|
+
const proc = await start(paneRef.role, 'database', 'database', spec, childEnv(spec.env), port)
|
|
543
|
+
const readySignal = signal ? AbortSignal.any([signal, proc.exitSignal]) : proc.exitSignal
|
|
544
|
+
try {
|
|
545
|
+
// raced too, so a ready check that ignores its signal can't hang the boot
|
|
546
|
+
await race([signal, proc.exitSignal], () => [(async () => database.ready({ port, signal: readySignal }))()])
|
|
547
|
+
} catch (err) {
|
|
548
|
+
if (!proc.exited && !signal?.aborted) throw err
|
|
549
|
+
}
|
|
550
|
+
// an abort wins: the teardown that follows one may be what stopped it
|
|
551
|
+
if (signal?.aborted) throw abortError()
|
|
552
|
+
if (proc.exited) {
|
|
553
|
+
throw withTail(new Error(`database exited before it was ready (${describeExit(proc.exit)})`), paneRef.role)
|
|
554
|
+
}
|
|
555
|
+
return database.handle({ paneRef, port })
|
|
556
|
+
},
|
|
557
|
+
|
|
558
|
+
async reserveServices({ paneRef, services }) {
|
|
559
|
+
const st = pane(paneRef.role)
|
|
560
|
+
const out = {}
|
|
561
|
+
for (const name of Object.keys(services)) {
|
|
562
|
+
const port = await allocatePort(st)
|
|
563
|
+
out[name] = { url: `http://${urlHost}:${port}`, port }
|
|
564
|
+
}
|
|
565
|
+
return out
|
|
566
|
+
},
|
|
567
|
+
|
|
568
|
+
async launchServices({ paneRef, services, env = {}, reserved, signal } = {}) {
|
|
569
|
+
const st = pane(paneRef.role)
|
|
570
|
+
if (st.stale) resetLogs(st) // the first launch since a teardown starts a new buffer
|
|
571
|
+
await stateDirReady()
|
|
572
|
+
for (const name of Object.keys(services)) {
|
|
573
|
+
const port = reserved?.[name]?.port
|
|
574
|
+
if (!port) throw new Error(`${name}: no reserved port (reserveServices first)`)
|
|
575
|
+
if (signal?.aborted) throw abortError()
|
|
576
|
+
const spec = command({ name, ref: services[name], port, env: env?.[name] ?? {}, paneRef })
|
|
577
|
+
await start(paneRef.role, 'service', name, spec, childEnv(env?.[name], spec.env), port)
|
|
578
|
+
}
|
|
579
|
+
},
|
|
580
|
+
|
|
581
|
+
async waitHealthy({ services, signal } = {}) {
|
|
582
|
+
for (const [name, svc] of Object.entries(services ?? {})) {
|
|
583
|
+
if (svc?.port) await waitOne(name, svc, signal)
|
|
584
|
+
}
|
|
585
|
+
},
|
|
586
|
+
|
|
587
|
+
// One buffer per pane, database and services together, each line
|
|
588
|
+
// prefixed [name]; `stage` doesn't narrow it.
|
|
589
|
+
async logs({ paneRef, lines = TAIL_LINES } = {}) {
|
|
590
|
+
return tail(paneRef?.role, lines)
|
|
591
|
+
},
|
|
592
|
+
|
|
593
|
+
// Orphans a previous run left behind. Entries of this instance's live
|
|
594
|
+
// processes are left alone; only the entries handled here are removed.
|
|
595
|
+
async sweep() {
|
|
596
|
+
await stateDirReady()
|
|
597
|
+
const boot = await bootId()
|
|
598
|
+
const done = []
|
|
599
|
+
for (const entry of await readPidfile()) {
|
|
600
|
+
if (ownedHere(entry)) continue
|
|
601
|
+
// Nothing outlives a reboot, so an entry from an earlier one names only
|
|
602
|
+
// dead processes, whatever holds its pid (or pgid) now.
|
|
603
|
+
if (boot && entry.bootId !== boot) {
|
|
604
|
+
done.push(entry)
|
|
605
|
+
continue
|
|
606
|
+
}
|
|
607
|
+
// Nothing to check it against: kept, until a reboot clears it.
|
|
608
|
+
if (!killablePid(entry.pid) || typeof entry.startedAt !== 'string') continue
|
|
609
|
+
try {
|
|
610
|
+
if (await sweepEntry(entry)) done.push(entry)
|
|
611
|
+
else log.warn(`[qa] sweep: process group ${entry.pid} (${entry.name}) survived SIGKILL; kept in ${pidfile}`)
|
|
612
|
+
} catch (err) {
|
|
613
|
+
log.warn(`[qa] sweep: skipped pid ${entry.pid}: ${err.message}`)
|
|
614
|
+
}
|
|
615
|
+
}
|
|
616
|
+
if (done.length) await editPidfile(without(done))
|
|
617
|
+
},
|
|
618
|
+
|
|
619
|
+
async teardown({ paneRef }) {
|
|
620
|
+
const st = panes.get(paneRef.role)
|
|
621
|
+
if (!st) return
|
|
622
|
+
st.stale = true
|
|
623
|
+
// services before the database they depend on
|
|
624
|
+
const order = [...st.procs.filter(p => p.kind !== 'database'), ...st.procs.filter(p => p.kind === 'database')]
|
|
625
|
+
const gone = []
|
|
626
|
+
for (const proc of order) {
|
|
627
|
+
if (killablePid(proc.pid)) {
|
|
628
|
+
const known = await ended(proc)
|
|
629
|
+
if (known === undefined) {
|
|
630
|
+
log.warn(`[qa] ${paneRef.role} ${proc.name}: could not check whether pid ${proc.pid} is still ours; not signalled, kept in ${pidfile}`)
|
|
631
|
+
continue
|
|
632
|
+
}
|
|
633
|
+
if (!known && !(await stopGroup(proc.pid, proc))) {
|
|
634
|
+
log.warn(`[qa] ${paneRef.role} ${proc.name}: process group ${proc.pid} survived SIGKILL; kept in ${pidfile} for the next sweep`)
|
|
635
|
+
continue
|
|
636
|
+
}
|
|
637
|
+
}
|
|
638
|
+
proc.gone = true
|
|
639
|
+
gone.push(proc)
|
|
640
|
+
}
|
|
641
|
+
st.procs = st.procs.filter(proc => !proc.gone)
|
|
642
|
+
for (const proc of gone) if (byPort.get(proc.port) === proc) byPort.delete(proc.port)
|
|
643
|
+
const held = new Set(st.procs.map(proc => proc.port))
|
|
644
|
+
for (const port of st.ports) {
|
|
645
|
+
if (held.has(port)) continue
|
|
646
|
+
allocated.delete(port)
|
|
647
|
+
st.ports.delete(port)
|
|
648
|
+
}
|
|
649
|
+
const recorded = gone.filter(proc => proc.pid !== null)
|
|
650
|
+
if (recorded.length) await editPidfile(without(recorded))
|
|
651
|
+
},
|
|
652
|
+
}
|
|
653
|
+
}
|