velocious 1.0.681 → 1.0.683

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,7 @@
1
1
  // @ts-check
2
2
 
3
3
  import { randomUUID } from "node:crypto"
4
+ import v8 from "node:v8"
4
5
  import timeout from "awaitery/build/timeout.js"
5
6
  import runJobPayload, { BackgroundJobPerformedFailure } from "./job-runner.js"
6
7
  import { boundedPooledRunnerInflightJobIds, isPooledChildShutdownReason, isPooledChildShutdownSignal } from "./pooled-runner-shutdown.js"
@@ -14,6 +15,10 @@ const BASE_PROCESS_TITLE = "velocious background-jobs-runner"
14
15
  const SHUTDOWN_OBSERVATION_SEND_TIMEOUT_MS = 100
15
16
  /** Stable identity of this pooled child process for the life of the process. */
16
17
  const childInstanceId = randomUUID()
18
+ /** Sampling cadence for the memory observation sampler. */
19
+ const MEMORY_OBSERVATION_INTERVAL_MS = 10000
20
+ /** Bound on in-flight job ids carried in a memory observation. */
21
+ const MEMORY_OBSERVATION_JOB_ID_LIMIT = 32
17
22
 
18
23
  setRunnerProcessTitle()
19
24
 
@@ -148,6 +153,93 @@ function sendChildAcceptance(type, payload, observedAtMs) {
148
153
  }
149
154
  }
150
155
 
156
+ /**
157
+ * Collects a bounded diagnostic snapshot of this child's memory state. The
158
+ * pooled child's stdio is ignored by the worker fork, so IPC is the only channel
159
+ * to the worker's log surface — this snapshot (sent periodically and on demand)
160
+ * is how a memory problem names itself in production. The V8 heap-stat
161
+ * breakdown (not a full heap snapshot, which would be far too expensive while a
162
+ * child runs 25 concurrent jobs) distinguishes a V8 heap-growth leak from
163
+ * external/array-buffer (native resource) growth, and the in-flight job ids
164
+ * tie the observation to the work that was running.
165
+ * @returns {{activeJobIds: string[], activeJobIdsTruncatedCount: number, childInstanceId: string, childPid: number, childUptimeMs: number, heapStatistics: ReturnType<typeof v8.getHeapStatistics>, jobCount: number, memoryUsage: ReturnType<typeof process.memoryUsage>, observedAtMs: number, rssBytes: number, type: "pooled-child-memory", uptimeMs: number}} - Bounded memory observation for this child.
166
+ */
167
+ function collectMemoryObservation() {
168
+ const jobIds = [...runningJobIds]
169
+ const activeJobIds = jobIds.slice(0, MEMORY_OBSERVATION_JOB_ID_LIMIT)
170
+ const memoryUsage = process.memoryUsage()
171
+
172
+ return {
173
+ activeJobIds,
174
+ activeJobIdsTruncatedCount: Math.max(0, jobIds.length - activeJobIds.length),
175
+ childInstanceId,
176
+ childPid: process.pid,
177
+ childUptimeMs: Math.floor(process.uptime() * 1000),
178
+ heapStatistics: v8.getHeapStatistics(),
179
+ jobCount: jobIds.length,
180
+ memoryUsage,
181
+ observedAtMs: Date.now(),
182
+ rssBytes: memoryUsage.rss,
183
+ type: "pooled-child-memory",
184
+ uptimeMs: Math.floor(process.uptime() * 1000)
185
+ }
186
+ }
187
+
188
+ /**
189
+ * Sends a memory observation to the worker. Sampling is cheap (process + V8
190
+ * heap stats) and only runs while jobs are in flight; a closed IPC channel is
191
+ * terminal for this child (the disconnect handler owns shutdown), so a failed
192
+ * send is swallowed.
193
+ * @returns {void}
194
+ */
195
+ function sendMemoryObservation() {
196
+ if (!process.send || runningJobIds.size === 0) return
197
+
198
+ try {
199
+ process.send(collectMemoryObservation())
200
+ } catch {
201
+ // The IPC channel is already gone; the disconnect handler owns shutdown.
202
+ }
203
+ }
204
+
205
+ /** @type {ReturnType<typeof setInterval> | undefined} */
206
+ let memoryObservationTimer
207
+
208
+ /**
209
+ * Checks whether an IPC value requests this child to send a memory observation.
210
+ * @param {ReturnType<typeof JSON.parse>} message - IPC message.
211
+ * @returns {message is {type: "memory-observation-request"}} - Whether the message requests one.
212
+ */
213
+ function isMemoryObservationRequestMessage(message) {
214
+ return message !== null
215
+ && typeof message === "object"
216
+ && message.type === "memory-observation-request"
217
+ }
218
+
219
+ /**
220
+ * Starts the periodic memory observation sampler if it is not already running.
221
+ * @returns {void}
222
+ */
223
+ function startMemoryObservationSampling() {
224
+ if (memoryObservationTimer || !process.send) return
225
+
226
+ memoryObservationTimer = setInterval(() => {
227
+ sendMemoryObservation()
228
+ }, MEMORY_OBSERVATION_INTERVAL_MS)
229
+ memoryObservationTimer.unref()
230
+ }
231
+
232
+ /**
233
+ * Stops the periodic memory observation sampler.
234
+ * @returns {void}
235
+ */
236
+ function stopMemoryObservationSampling() {
237
+ if (!memoryObservationTimer) return
238
+
239
+ clearInterval(memoryObservationTimer)
240
+ memoryObservationTimer = undefined
241
+ }
242
+
151
243
  /**
152
244
  * Checks whether an IPC value is a runnable pooled job message.
153
245
  * @param {ReturnType<typeof JSON.parse>} message - IPC message.
@@ -238,6 +330,10 @@ async function runJob(payload, sharedTransactionBroker) {
238
330
  } finally {
239
331
  runningJobIds.delete(payload.id)
240
332
  updateProcessTitle()
333
+
334
+ if (runningJobIds.size === 0) {
335
+ stopMemoryObservationSampling()
336
+ }
241
337
  }
242
338
  }
243
339
 
@@ -257,11 +353,17 @@ function handleMessage(message) {
257
353
  return
258
354
  }
259
355
 
356
+ if (isMemoryObservationRequestMessage(message)) {
357
+ sendMemoryObservation()
358
+ return
359
+ }
360
+
260
361
  if (!isJobMessage(message) || runningJobIds.has(message.payload.id)) return
261
362
 
262
363
  runningJobIds.add(message.payload.id)
263
364
  updateProcessTitle()
264
365
  sendChildAcceptance("job-received", message.payload, Date.now())
366
+ startMemoryObservationSampling()
265
367
  void runJob(message.payload, message.sharedTransactionBroker || {expected: false})
266
368
  }
267
369
 
@@ -73,6 +73,27 @@
73
73
  * @property {number | null} shutdownRequestedAtMs - Parent request timestamp when supplied over IPC.
74
74
  * @property {import("node:child_process").ChildProcess["signalCode"]} signal - Requested or observed signal when available.
75
75
  */
76
+ /**
77
+ * Bounded pooled-child memory observation sent over the child IPC channel.
78
+ * The pooled child's stdio is ignored by the worker fork, so this observation
79
+ * (sent periodically while jobs run, plus on demand) is how a memory problem
80
+ * in a running child names itself. `heapStatistics` and `memoryUsage`
81
+ * distinguish V8-heap growth from external/array-buffer (native resource)
82
+ * growth; `activeJobIds` ties the sample to the work that was in flight.
83
+ * @typedef {object} PooledChildMemoryObservation
84
+ * @property {string[]} activeJobIds - In-flight job ids, bounded.
85
+ * @property {number} activeJobIdsTruncatedCount - In-flight job ids omitted by the bound.
86
+ * @property {string} childInstanceId - Stable identity reported by the child.
87
+ * @property {number} childPid - Child process id.
88
+ * @property {number} childUptimeMs - Child process uptime in ms.
89
+ * @property {ReturnType<typeof import("node:v8").getHeapStatistics>} heapStatistics - V8 heap-stat breakdown at the sample.
90
+ * @property {number} jobCount - In-flight job count.
91
+ * @property {ReturnType<typeof import("node:process").memoryUsage>} memoryUsage - Process memory breakdown at the sample.
92
+ * @property {number} observedAtMs - Epoch ms the child sampled.
93
+ * @property {number} rssBytes - Resident set size in bytes at the sample.
94
+ * @property {"pooled-child-memory"} type - Discriminator.
95
+ * @property {number} uptimeMs - Process uptime in ms.
96
+ */
76
97
  /**
77
98
  * @typedef {object} LocalBackgroundJobsClock
78
99
  * @property {() => number} now - Current epoch milliseconds.
@@ -41,6 +41,7 @@ import { POOLED_RUNNER_INFLIGHT_JOB_ID_LIMIT, boundedPooledRunnerInflightJobIds,
41
41
  * @property {boolean} retiring - Whether this child is draining before retirement.
42
42
  * @property {boolean} [started] - Whether the child completed its startup handshake.
43
43
  * @property {boolean} [settling] - Whether failure handling already owns this child.
44
+ * @property {import("./types.js").PooledChildMemoryObservation} [lastMemoryObservation] - Latest memory observation received from this child.
44
45
  * @property {number} [ipcDisconnectedAtMs] - Parent observation of IPC disconnect.
45
46
  * @property {import("./types.js").PooledChildShutdownObservation} [shutdownObservation] - Child observation sent before teardown.
46
47
  * @property {import("./types.js").PooledChildShutdownReason} [shutdownReason] - Exact parent-requested shutdown reason.
@@ -132,6 +133,38 @@ function isChildShutdownObservationMessage(message) {
132
133
  && isPooledChildShutdownSignal(record.signal)
133
134
  }
134
135
 
136
+ /**
137
+ * Checks whether an IPC value is a pooled child's bounded memory observation.
138
+ * The pooled child's stdio is ignored by the worker fork, so this observation
139
+ * (periodic while jobs run, plus on demand) is how a memory problem in a
140
+ * running child names itself. The worker logs it (its stderr reaches the prod
141
+ * log) and forwards it to the optional `onPooledRunnerMemoryObservation` hook.
142
+ * @param {ReturnType<typeof JSON.parse>} message - IPC message.
143
+ * @returns {message is import("./types.js").PooledChildMemoryObservation} - Whether this is a valid memory observation.
144
+ */
145
+ function isPooledChildMemoryObservationMessage(message) {
146
+ if (!message || typeof message !== "object") return false
147
+ const record = /** @type {Record<string, ReturnType<typeof JSON.parse>>} */ (message)
148
+ const heapStatistics = /** @type {Record<string, ReturnType<typeof JSON.parse>> | undefined} */ (record.heapStatistics)
149
+ const memoryUsage = /** @type {Record<string, ReturnType<typeof JSON.parse>> | undefined} */ (record.memoryUsage)
150
+
151
+ return record.type === "pooled-child-memory"
152
+ && typeof record.childInstanceId === "string"
153
+ && Number.isInteger(record.childPid)
154
+ && typeof record.rssBytes === "number"
155
+ && Number.isFinite(record.rssBytes)
156
+ && Number.isInteger(record.jobCount)
157
+ && record.jobCount >= 0
158
+ && Array.isArray(record.activeJobIds)
159
+ && record.activeJobIds.every((jobId) => typeof jobId === "string")
160
+ && Number.isInteger(record.activeJobIdsTruncatedCount)
161
+ && record.activeJobIdsTruncatedCount >= 0
162
+ && typeof record.observedAtMs === "number"
163
+ && Number.isFinite(record.observedAtMs)
164
+ && heapStatistics !== undefined && typeof heapStatistics === "object"
165
+ && memoryUsage !== undefined && typeof memoryUsage === "object"
166
+ }
167
+
135
168
  /**
136
169
  * Normalizes a candidate pooled-runner resource limit.
137
170
  * @param {number | undefined} value - Candidate positive number.
@@ -166,8 +199,9 @@ export default class BackgroundJobsWorker {
166
199
  * @param {() => void | Promise<void>} [args.onStopped] - Lifecycle hook invoked after the worker finishes stopping.
167
200
  * @param {() => void} [args.onGenerationAccepted] - Explicit generation-acceptance observation hook.
168
201
  * @param {() => void} [args.onRetireMessage] - Explicit retire-message observation hook.
202
+ * @param {(observation: import("./types.js").PooledChildMemoryObservation) => void | Promise<void>} [args.onPooledRunnerMemoryObservation] - Explicit pooled-child memory observation hook. Every validated observation (periodic while a child has in-flight jobs, plus on demand) is forwarded here, in addition to the worker's compact stderr log line, so an application can route memory diagnostics (e.g. to a bug reporter) without parsing logs.
169
203
  */
170
- constructor({configuration, host, port, generationId, workerInstanceId, maxConcurrentForkedJobs, maxConcurrentInlineJobs, pooledRunnerCount, pooledRunnerConcurrency, pooledRunnerMaxJobs, pooledRunnerMaxRssBytes, pooledRunnerMaxLifetimeMs, forkedChildSigkillGraceMs, heartbeatIntervalMs, generationHandshakeTimeoutMs = DEFAULT_GENERATION_HANDSHAKE_TIMEOUT_MS, reconnectDelayMs = 1000, jobTimeoutMs, closeDatabaseConnectionsOnStop = true, onStopped, onGenerationAccepted, onRetireMessage} = {}) {
204
+ constructor({configuration, host, port, generationId, workerInstanceId, maxConcurrentForkedJobs, maxConcurrentInlineJobs, pooledRunnerCount, pooledRunnerConcurrency, pooledRunnerMaxJobs, pooledRunnerMaxRssBytes, pooledRunnerMaxLifetimeMs, forkedChildSigkillGraceMs, heartbeatIntervalMs, generationHandshakeTimeoutMs = DEFAULT_GENERATION_HANDSHAKE_TIMEOUT_MS, reconnectDelayMs = 1000, jobTimeoutMs, closeDatabaseConnectionsOnStop = true, onStopped, onGenerationAccepted, onRetireMessage, onPooledRunnerMemoryObservation} = {}) {
171
205
  /**
172
206
  * Narrows the runtime value to the documented type.
173
207
  * @type {Promise<import("../configuration.js").default>} */
@@ -186,6 +220,7 @@ export default class BackgroundJobsWorker {
186
220
  this.onStopped = onStopped
187
221
  this.onGenerationAccepted = onGenerationAccepted
188
222
  this.onRetireMessage = onRetireMessage
223
+ this.onPooledRunnerMemoryObservation = onPooledRunnerMemoryObservation
189
224
  /**
190
225
  * Constructor override for the inline-job concurrency cap. When unset
191
226
  * the cap is read from `configuration.getBackgroundJobsConfig()` in
@@ -322,6 +357,28 @@ export default class BackgroundJobsWorker {
322
357
  // Monotonic dispatch counter for round-robin child selection: each dispatch stamps
323
358
  // the chosen child, and selection prefers the child dispatched least recently.
324
359
  this._pooledDispatchSeq = 0
360
+ // Waiters blocked in _runPooledJob because the pool is at its hard cap: a job may
361
+ // not spawn a child while total live children (working + draining) is at the cap.
362
+ /** @type {Set<() => void>} */
363
+ this._pooledSlotWaiters = new Set()
364
+ /** @type {ReturnType<typeof setInterval> | undefined} - Safety poll that re-checks the slot condition. */
365
+ this._pooledSlotWaitTimer = undefined
366
+ }
367
+
368
+ /** Starts the slot-waiter safety poll if it is not already running. */
369
+ _startPooledSlotWaitPoll() {
370
+ if (this._pooledSlotWaitTimer) return
371
+
372
+ this._pooledSlotWaitTimer = setInterval(() => this._wakePooledSlotWaiters(), 50)
373
+ this._pooledSlotWaitTimer.unref()
374
+ }
375
+
376
+ /** Stops the slot-waiter safety poll once no waiter is registered. */
377
+ _stopPooledSlotWaitPollIfIdle() {
378
+ if (this._pooledSlotWaiters.size > 0 || !this._pooledSlotWaitTimer) return
379
+
380
+ clearInterval(this._pooledSlotWaitTimer)
381
+ this._pooledSlotWaitTimer = undefined
325
382
  }
326
383
 
327
384
  /**
@@ -1079,27 +1136,72 @@ export default class BackgroundJobsWorker {
1079
1136
  }
1080
1137
  }
1081
1138
 
1139
+ /**
1140
+ * Hard cap on total live pooled children (working + draining): children the
1141
+ * pool may still spawn. Counting the whole live set — draining children
1142
+ * included — is what bounds pool memory: a draining child still holds its
1143
+ * RSS until its last in-flight job finishes, so it must occupy a cap slot.
1144
+ * @returns {number} - Number of children the pool may still spawn.
1145
+ */
1146
+ _spawnablePooledChildren() {
1147
+ return Math.max(0, this.pooledRunnerCount - this.pooledChildren.size)
1148
+ }
1149
+
1150
+ /**
1151
+ * Resolves once a non-retiring pooled child has a free concurrency slot or the
1152
+ * pool may spawn a new one. Pooled jobs admitted while the pool is at its
1153
+ * hard cap wait here instead of spawning an over-capacity child; the wake
1154
+ * points are the only capacity-freeing transitions (a job outcome and a child
1155
+ * exit), so no polling is needed.
1156
+ * @returns {Promise<void>} - Resolves when a slot is available.
1157
+ */
1158
+ _waitPooledSlot() {
1159
+ this._startPooledSlotWaitPoll()
1160
+
1161
+ return new Promise((resolve) => {
1162
+ const waiter = () => {
1163
+ this._pooledSlotWaiters.delete(waiter)
1164
+ this._stopPooledSlotWaitPollIfIdle()
1165
+ resolve()
1166
+ }
1167
+
1168
+ this._pooledSlotWaiters.add(waiter)
1169
+ })
1170
+ }
1171
+
1172
+ /**
1173
+ * Resolves every registered waiter; waiters re-check the slot condition
1174
+ * themselves and only proceed when it holds.
1175
+ * @returns {void}
1176
+ */
1177
+ _wakePooledSlotWaiters() {
1178
+ if (this._pooledSlotWaiters.size === 0) return
1179
+
1180
+ for (const resolve of [...this._pooledSlotWaiters]) resolve()
1181
+ }
1182
+
1082
1183
  /**
1083
1184
  * Free pooled slots across the pool: open slots in non-retiring children plus
1084
- * the slots we could add by spawning more children up to `pooledRunnerCount`.
1085
- * Retiring children (draining before replacement) never contribute capacity.
1185
+ * the slots we could add by spawning more children up to the hard cap on total
1186
+ * live children. Retiring children (draining before replacement) never
1187
+ * contribute capacity, and they count against the cap: while one is still
1188
+ * draining, no replacement is advertised (or spawned) — the pool advertises
1189
+ * exactly what it can serve instead of phantom capacity.
1086
1190
  * @returns {number} - Number of pooled jobs the worker can accept right now.
1087
1191
  */
1088
1192
  _availablePooledSlots() {
1089
1193
  let openInExisting = 0
1090
- let nonRetiringChildren = 0
1091
1194
  let queuedReservations = 0
1092
1195
 
1093
1196
  for (const child of this.pooledChildren) {
1094
1197
  const state = this.pooledChildStates.get(child)
1095
1198
  if (!state || state.retiring) continue
1096
- nonRetiringChildren += 1
1097
1199
  openInExisting += this.pooledRunnerConcurrency - state.inflight.size
1098
1200
  }
1099
1201
 
1100
1202
  for (const queue of this.pooledJobQueues.values()) queuedReservations += queue.length
1101
1203
 
1102
- const spawnableChildren = Math.max(0, this.pooledRunnerCount - nonRetiringChildren)
1204
+ const spawnableChildren = Math.max(0, this.pooledRunnerCount - this.pooledChildren.size)
1103
1205
 
1104
1206
  return Math.max(0, openInExisting + spawnableChildren * this.pooledRunnerConcurrency - queuedReservations)
1105
1207
  }
@@ -1109,11 +1211,32 @@ export default class BackgroundJobsWorker {
1109
1211
  * new child when every non-retiring child is full and the pool is below
1110
1212
  * `pooledRunnerCount`. Each child runs up to `pooledRunnerConcurrency` jobs at
1111
1213
  * once on its own event loop.
1214
+ *
1215
+ * When the pool is already at its hard cap (total live children, draining
1216
+ * included), the job waits for a slot instead of spawning: that is what keeps
1217
+ * the live-child count — and therefore the pool's total RSS — bounded. The
1218
+ * wait resolves on the only two capacity-freeing transitions (a job outcome,
1219
+ * a child exit); a safety poll covers anything missed.
1112
1220
  * @param {import("./types.js").BackgroundJobPayload & {id: string}} payload - Job payload.
1113
1221
  * @returns {Promise<void>} - Resolves after the durable report.
1114
1222
  */
1115
- _runPooledJob(payload) {
1116
- const child = this._selectPooledChild() || this._createPooledChild()
1223
+ async _runPooledJob(payload) {
1224
+ // At the hard cap (no free slot, no spawnable child) the job waits for a
1225
+ // slot instead of spawning an over-capacity child — that is what bounds
1226
+ // the live-child count and the pool's total RSS.
1227
+ let child = this._selectPooledChild()
1228
+ while (!child) {
1229
+ if (this._spawnablePooledChildren() === 0) {
1230
+ // Shutdown: main no longer dispatches and no slot will ever free —
1231
+ // stop waiting so the tracked job can settle and the drain completes.
1232
+ if (this.shouldStop) return
1233
+ await this._waitPooledSlot()
1234
+ child = this._selectPooledChild()
1235
+ continue
1236
+ }
1237
+ child = this._selectPooledChild() || this._createPooledChild()
1238
+ }
1239
+
1117
1240
  const state = this.pooledChildStates.get(child)
1118
1241
  if (!state) throw new Error("Pooled runner state missing")
1119
1242
 
@@ -1232,12 +1355,16 @@ export default class BackgroundJobsWorker {
1232
1355
  }
1233
1356
 
1234
1357
  /**
1235
- * Creates a reusable pooled child.
1236
- * @returns {import("node:child_process").ChildProcess} - New pooled child.
1358
+ * Creates a reusable pooled child, enforcing the hard cap on total live
1359
+ * children (working + draining). Returns undefined when the cap is already
1360
+ * met — the only way a new child may exist is a slot being open, so the
1361
+ * caller re-checks and waits again.
1362
+ * @returns {import("node:child_process").ChildProcess | undefined} - The new child, or undefined when the pool is at its cap.
1237
1363
  */
1238
1364
  _createPooledChild() {
1239
1365
  const configuration = this.configuration
1240
1366
  if (!configuration) throw new Error("Background jobs worker configuration not initialized")
1367
+ if (this.pooledChildren.size >= this.pooledRunnerCount) return undefined
1241
1368
  const child = fork(POOLED_RUNNER_ENTRY_PATH, [], {
1242
1369
  cwd: configuration.getDirectory(), execArgv: [], stdio: ["ignore", "ignore", "ignore", "ipc"],
1243
1370
  env: Object.assign({}, process.env, this._childBackgroundJobsEnvironment())
@@ -1301,6 +1428,10 @@ export default class BackgroundJobsWorker {
1301
1428
  this._reportChildAccepted(message)
1302
1429
  return
1303
1430
  }
1431
+ if (isPooledChildMemoryObservationMessage(message)) {
1432
+ this._handlePooledChildMemoryObservation({child, message})
1433
+ return
1434
+ }
1304
1435
  if (record.type !== "job-outcome" || !state || state.settling || typeof record.jobId !== "string") return
1305
1436
  state.started = true
1306
1437
  const entry = state.inflight.get(record.jobId)
@@ -1332,6 +1463,9 @@ export default class BackgroundJobsWorker {
1332
1463
  this._beginRetirePooledChild(child)
1333
1464
  }
1334
1465
  this._terminateIfDrained(child)
1466
+ // A job outcome frees a concurrency slot and may drain a retiring child —
1467
+ // the only two transitions that free capacity for waiters at the hard cap.
1468
+ this._wakePooledSlotWaiters()
1335
1469
  }
1336
1470
 
1337
1471
  /**
@@ -1362,11 +1496,85 @@ export default class BackgroundJobsWorker {
1362
1496
  }
1363
1497
 
1364
1498
  /**
1365
- * Marks a pooled child for retirement and eagerly spawns a single replacement
1366
- * (1-for-1) so its capacity is restored immediately without waiting for it to
1367
- * finish draining. The retiring child stops receiving new jobs and is
1368
- * terminated only once its in-flight set drains, so a long-running job (e.g. a
1369
- * build) is never cut off.
1499
+ * Handles a pooled child's bounded memory observation. The pooled child's
1500
+ * stdio is ignored by the worker fork, so this IPC observation is how a
1501
+ * memory problem in a running child names itself. The worker (a) records the
1502
+ * latest observation on the child's state for later correlation, (b) logs one
1503
+ * compact line to its own stderr (which reaches the prod log, unlike the
1504
+ * child's ignored stdio), and (c) forwards the full observation to the
1505
+ * optional `onPooledRunnerMemoryObservation` hook so an application can route
1506
+ * it (e.g. to a bug reporter) without parsing logs. The heap-stat breakdown
1507
+ * distinguishes V8-heap growth from external/array-buffer (native) growth. A
1508
+ * hook failure is swallowed — diagnostics must never take down the worker or
1509
+ * fail the jobs running on that child.
1510
+ * @param {object} args - Message details.
1511
+ * @param {import("node:child_process").ChildProcess} args.child - Pooled child.
1512
+ * @param {import("./types.js").PooledChildMemoryObservation} args.message - Validated memory observation.
1513
+ * @returns {void}
1514
+ */
1515
+ _handlePooledChildMemoryObservation({child, message}) {
1516
+ const state = this.pooledChildStates.get(child)
1517
+ if (state) state.lastMemoryObservation = message
1518
+
1519
+ const heap = message.heapStatistics
1520
+ console.error(
1521
+ JSON.stringify({
1522
+ event: "pooled-child-memory",
1523
+ childInstanceId: state?.childInstanceId ?? message.childInstanceId,
1524
+ childPid: message.childPid,
1525
+ childUptimeS: Math.round(message.childUptimeMs / 1000),
1526
+ rssMb: Math.round(message.rssBytes / (1024 * 1024)),
1527
+ heapUsedMb: Math.round(heap.used_heap_size / (1024 * 1024)),
1528
+ heapTotalMb: Math.round(heap.total_heap_size / (1024 * 1024)),
1529
+ heapLimitMb: Math.round(heap.heap_size_limit / (1024 * 1024)),
1530
+ externalMb: Math.round(message.memoryUsage.external / (1024 * 1024)),
1531
+ arrayBuffersMb: Math.round(message.memoryUsage.arrayBuffers / (1024 * 1024)),
1532
+ jobs: message.jobCount,
1533
+ activeJobIds: message.activeJobIds
1534
+ })
1535
+ )
1536
+
1537
+ if (this.onPooledRunnerMemoryObservation) {
1538
+ try {
1539
+ const result = this.onPooledRunnerMemoryObservation(message)
1540
+ if (result && typeof result.catch === "function") result.catch((error) => {
1541
+ console.error("Pooled runner memory observation hook failed:", error)
1542
+ })
1543
+ } catch (error) {
1544
+ console.error("Pooled runner memory observation hook failed:", error)
1545
+ }
1546
+ }
1547
+ }
1548
+
1549
+ /**
1550
+ * Requests an immediate memory observation from one pooled child. The child
1551
+ * replies over IPC with its current snapshot (the same shape as the periodic
1552
+ * sampler), which the worker records, logs, and forwards to the
1553
+ * `onPooledRunnerMemoryObservation` hook. Use this to pull a snapshot on
1554
+ * suspicion (e.g. after an OOM report) without waiting for the next periodic
1555
+ * sample. A no-op when the child is gone or its IPC channel is closed.
1556
+ * @param {import("node:child_process").ChildProcess} child - Pooled child to sample.
1557
+ * @returns {void}
1558
+ */
1559
+ requestPooledChildMemoryObservation(child) {
1560
+ if (!child.connected) return
1561
+
1562
+ try {
1563
+ child.send({type: "memory-observation-request"})
1564
+ } catch {
1565
+ // The IPC channel is already gone; the disconnect/exit handler owns teardown.
1566
+ }
1567
+ }
1568
+
1569
+ /**
1570
+ * Marks a pooled child for retirement and — when the pool is below its hard
1571
+ * cap — eagerly spawns a single replacement (1-for-1) so its capacity is
1572
+ * restored immediately without waiting for it to finish draining. The
1573
+ * replacement spawn is gated by the cap (the retiring child still counts as
1574
+ * live until it exits), so a full pool simply defers the replacement to the
1575
+ * retiring child's drain instead of spawning over capacity. The retiring
1576
+ * child stops receiving new jobs and is terminated only once its in-flight
1577
+ * set drains, so a long-running job (e.g. a build) is never cut off.
1370
1578
  * @param {import("node:child_process").ChildProcess} child - Child to retire.
1371
1579
  * @returns {void}
1372
1580
  */
@@ -1376,7 +1584,9 @@ export default class BackgroundJobsWorker {
1376
1584
 
1377
1585
  state.retiring = true
1378
1586
  // Best-effort pre-warm: skip when stopping (no new work) or before the
1379
- // worker is initialized (no configuration to fork a child from).
1587
+ // worker is initialized (no configuration to fork a child from). The cap
1588
+ // inside _createPooledChild refuses the spawn while the pool is full, in
1589
+ // which case the replacement is deferred to the drain path.
1380
1590
  if (!this.shouldStop && this.configuration) this._createPooledChild()
1381
1591
  }
1382
1592
 
@@ -1394,6 +1604,9 @@ export default class BackgroundJobsWorker {
1394
1604
 
1395
1605
  /**
1396
1606
  * Retires a drained pooled child (removes it from tracking, then SIGTERMs it).
1607
+ * Because the hard cap counts live children, the exit of this child frees a
1608
+ * slot: any deferred replacement (the pool was full when the child retired)
1609
+ * is spawned now, and capacity is re-advertised so main can dispatch into it.
1397
1610
  * @param {import("node:child_process").ChildProcess} child - Child process to retire.
1398
1611
  * @returns {void}
1399
1612
  */
@@ -1503,6 +1716,9 @@ export default class BackgroundJobsWorker {
1503
1716
  }
1504
1717
  this.pooledChildren.delete(child)
1505
1718
  this.inflightProcessChildren.delete(child)
1719
+ // Child exit frees a hard-cap slot even while its in-flight set is still
1720
+ // being reported — wake waiters now; their reports settle independently.
1721
+ this._wakePooledSlotWaiters()
1506
1722
 
1507
1723
  const entries = state ? [...state.inflight.values()] : []
1508
1724
  const runnerFailure = state