velocious 1.0.681 → 1.0.683
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/build/background-jobs/pooled-runner-child.js +102 -0
- package/build/background-jobs/types.js +21 -0
- package/build/background-jobs/worker.js +232 -16
- package/build/src/background-jobs/pooled-runner-child.js +95 -1
- package/build/src/background-jobs/types.d.ts +71 -0
- package/build/src/background-jobs/types.d.ts.map +1 -1
- package/build/src/background-jobs/types.js +22 -1
- package/build/src/background-jobs/worker.d.ts +99 -11
- package/build/src/background-jobs/worker.d.ts.map +1 -1
- package/build/src/background-jobs/worker.js +225 -17
- package/package.json +1 -1
- package/src/background-jobs/pooled-runner-child.js +102 -0
- package/src/background-jobs/types.js +21 -0
- package/src/background-jobs/worker.js +232 -16
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
// @ts-check
|
|
2
2
|
|
|
3
3
|
import { randomUUID } from "node:crypto"
|
|
4
|
+
import v8 from "node:v8"
|
|
4
5
|
import timeout from "awaitery/build/timeout.js"
|
|
5
6
|
import runJobPayload, { BackgroundJobPerformedFailure } from "./job-runner.js"
|
|
6
7
|
import { boundedPooledRunnerInflightJobIds, isPooledChildShutdownReason, isPooledChildShutdownSignal } from "./pooled-runner-shutdown.js"
|
|
@@ -14,6 +15,10 @@ const BASE_PROCESS_TITLE = "velocious background-jobs-runner"
|
|
|
14
15
|
const SHUTDOWN_OBSERVATION_SEND_TIMEOUT_MS = 100
|
|
15
16
|
/** Stable identity of this pooled child process for the life of the process. */
|
|
16
17
|
const childInstanceId = randomUUID()
|
|
18
|
+
/** Sampling cadence for the memory observation sampler. */
|
|
19
|
+
const MEMORY_OBSERVATION_INTERVAL_MS = 10000
|
|
20
|
+
/** Bound on in-flight job ids carried in a memory observation. */
|
|
21
|
+
const MEMORY_OBSERVATION_JOB_ID_LIMIT = 32
|
|
17
22
|
|
|
18
23
|
setRunnerProcessTitle()
|
|
19
24
|
|
|
@@ -148,6 +153,93 @@ function sendChildAcceptance(type, payload, observedAtMs) {
|
|
|
148
153
|
}
|
|
149
154
|
}
|
|
150
155
|
|
|
156
|
+
/**
|
|
157
|
+
* Collects a bounded diagnostic snapshot of this child's memory state. The
|
|
158
|
+
* pooled child's stdio is ignored by the worker fork, so IPC is the only channel
|
|
159
|
+
* to the worker's log surface — this snapshot (sent periodically and on demand)
|
|
160
|
+
* is how a memory problem names itself in production. The V8 heap-stat
|
|
161
|
+
* breakdown (not a full heap snapshot, which would be far too expensive while a
|
|
162
|
+
* child runs 25 concurrent jobs) distinguishes a V8 heap-growth leak from
|
|
163
|
+
* external/array-buffer (native resource) growth, and the in-flight job ids
|
|
164
|
+
* tie the observation to the work that was running.
|
|
165
|
+
* @returns {{activeJobIds: string[], activeJobIdsTruncatedCount: number, childInstanceId: string, childPid: number, childUptimeMs: number, heapStatistics: ReturnType<typeof v8.getHeapStatistics>, jobCount: number, memoryUsage: ReturnType<typeof process.memoryUsage>, observedAtMs: number, rssBytes: number, type: "pooled-child-memory", uptimeMs: number}} - Bounded memory observation for this child.
|
|
166
|
+
*/
|
|
167
|
+
function collectMemoryObservation() {
|
|
168
|
+
const jobIds = [...runningJobIds]
|
|
169
|
+
const activeJobIds = jobIds.slice(0, MEMORY_OBSERVATION_JOB_ID_LIMIT)
|
|
170
|
+
const memoryUsage = process.memoryUsage()
|
|
171
|
+
|
|
172
|
+
return {
|
|
173
|
+
activeJobIds,
|
|
174
|
+
activeJobIdsTruncatedCount: Math.max(0, jobIds.length - activeJobIds.length),
|
|
175
|
+
childInstanceId,
|
|
176
|
+
childPid: process.pid,
|
|
177
|
+
childUptimeMs: Math.floor(process.uptime() * 1000),
|
|
178
|
+
heapStatistics: v8.getHeapStatistics(),
|
|
179
|
+
jobCount: jobIds.length,
|
|
180
|
+
memoryUsage,
|
|
181
|
+
observedAtMs: Date.now(),
|
|
182
|
+
rssBytes: memoryUsage.rss,
|
|
183
|
+
type: "pooled-child-memory",
|
|
184
|
+
uptimeMs: Math.floor(process.uptime() * 1000)
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* Sends a memory observation to the worker. Sampling is cheap (process + V8
|
|
190
|
+
* heap stats) and only runs while jobs are in flight; a closed IPC channel is
|
|
191
|
+
* terminal for this child (the disconnect handler owns shutdown), so a failed
|
|
192
|
+
* send is swallowed.
|
|
193
|
+
* @returns {void}
|
|
194
|
+
*/
|
|
195
|
+
function sendMemoryObservation() {
|
|
196
|
+
if (!process.send || runningJobIds.size === 0) return
|
|
197
|
+
|
|
198
|
+
try {
|
|
199
|
+
process.send(collectMemoryObservation())
|
|
200
|
+
} catch {
|
|
201
|
+
// The IPC channel is already gone; the disconnect handler owns shutdown.
|
|
202
|
+
}
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/** @type {ReturnType<typeof setInterval> | undefined} */
|
|
206
|
+
let memoryObservationTimer
|
|
207
|
+
|
|
208
|
+
/**
|
|
209
|
+
* Checks whether an IPC value requests this child to send a memory observation.
|
|
210
|
+
* @param {ReturnType<typeof JSON.parse>} message - IPC message.
|
|
211
|
+
* @returns {message is {type: "memory-observation-request"}} - Whether the message requests one.
|
|
212
|
+
*/
|
|
213
|
+
function isMemoryObservationRequestMessage(message) {
|
|
214
|
+
return message !== null
|
|
215
|
+
&& typeof message === "object"
|
|
216
|
+
&& message.type === "memory-observation-request"
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/**
|
|
220
|
+
* Starts the periodic memory observation sampler if it is not already running.
|
|
221
|
+
* @returns {void}
|
|
222
|
+
*/
|
|
223
|
+
function startMemoryObservationSampling() {
|
|
224
|
+
if (memoryObservationTimer || !process.send) return
|
|
225
|
+
|
|
226
|
+
memoryObservationTimer = setInterval(() => {
|
|
227
|
+
sendMemoryObservation()
|
|
228
|
+
}, MEMORY_OBSERVATION_INTERVAL_MS)
|
|
229
|
+
memoryObservationTimer.unref()
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/**
|
|
233
|
+
* Stops the periodic memory observation sampler.
|
|
234
|
+
* @returns {void}
|
|
235
|
+
*/
|
|
236
|
+
function stopMemoryObservationSampling() {
|
|
237
|
+
if (!memoryObservationTimer) return
|
|
238
|
+
|
|
239
|
+
clearInterval(memoryObservationTimer)
|
|
240
|
+
memoryObservationTimer = undefined
|
|
241
|
+
}
|
|
242
|
+
|
|
151
243
|
/**
|
|
152
244
|
* Checks whether an IPC value is a runnable pooled job message.
|
|
153
245
|
* @param {ReturnType<typeof JSON.parse>} message - IPC message.
|
|
@@ -238,6 +330,10 @@ async function runJob(payload, sharedTransactionBroker) {
|
|
|
238
330
|
} finally {
|
|
239
331
|
runningJobIds.delete(payload.id)
|
|
240
332
|
updateProcessTitle()
|
|
333
|
+
|
|
334
|
+
if (runningJobIds.size === 0) {
|
|
335
|
+
stopMemoryObservationSampling()
|
|
336
|
+
}
|
|
241
337
|
}
|
|
242
338
|
}
|
|
243
339
|
|
|
@@ -257,11 +353,17 @@ function handleMessage(message) {
|
|
|
257
353
|
return
|
|
258
354
|
}
|
|
259
355
|
|
|
356
|
+
if (isMemoryObservationRequestMessage(message)) {
|
|
357
|
+
sendMemoryObservation()
|
|
358
|
+
return
|
|
359
|
+
}
|
|
360
|
+
|
|
260
361
|
if (!isJobMessage(message) || runningJobIds.has(message.payload.id)) return
|
|
261
362
|
|
|
262
363
|
runningJobIds.add(message.payload.id)
|
|
263
364
|
updateProcessTitle()
|
|
264
365
|
sendChildAcceptance("job-received", message.payload, Date.now())
|
|
366
|
+
startMemoryObservationSampling()
|
|
265
367
|
void runJob(message.payload, message.sharedTransactionBroker || {expected: false})
|
|
266
368
|
}
|
|
267
369
|
|
|
@@ -73,6 +73,27 @@
|
|
|
73
73
|
* @property {number | null} shutdownRequestedAtMs - Parent request timestamp when supplied over IPC.
|
|
74
74
|
* @property {import("node:child_process").ChildProcess["signalCode"]} signal - Requested or observed signal when available.
|
|
75
75
|
*/
|
|
76
|
+
/**
|
|
77
|
+
* Bounded pooled-child memory observation sent over the child IPC channel.
|
|
78
|
+
* The pooled child's stdio is ignored by the worker fork, so this observation
|
|
79
|
+
* (sent periodically while jobs run, plus on demand) is how a memory problem
|
|
80
|
+
* in a running child names itself. `heapStatistics` and `memoryUsage`
|
|
81
|
+
* distinguish V8-heap growth from external/array-buffer (native resource)
|
|
82
|
+
* growth; `activeJobIds` ties the sample to the work that was in flight.
|
|
83
|
+
* @typedef {object} PooledChildMemoryObservation
|
|
84
|
+
* @property {string[]} activeJobIds - In-flight job ids, bounded.
|
|
85
|
+
* @property {number} activeJobIdsTruncatedCount - In-flight job ids omitted by the bound.
|
|
86
|
+
* @property {string} childInstanceId - Stable identity reported by the child.
|
|
87
|
+
* @property {number} childPid - Child process id.
|
|
88
|
+
* @property {number} childUptimeMs - Child process uptime in ms.
|
|
89
|
+
* @property {ReturnType<typeof import("node:v8").getHeapStatistics>} heapStatistics - V8 heap-stat breakdown at the sample.
|
|
90
|
+
* @property {number} jobCount - In-flight job count.
|
|
91
|
+
* @property {ReturnType<typeof import("node:process").memoryUsage>} memoryUsage - Process memory breakdown at the sample.
|
|
92
|
+
* @property {number} observedAtMs - Epoch ms the child sampled.
|
|
93
|
+
* @property {number} rssBytes - Resident set size in bytes at the sample.
|
|
94
|
+
* @property {"pooled-child-memory"} type - Discriminator.
|
|
95
|
+
* @property {number} uptimeMs - Process uptime in ms.
|
|
96
|
+
*/
|
|
76
97
|
/**
|
|
77
98
|
* @typedef {object} LocalBackgroundJobsClock
|
|
78
99
|
* @property {() => number} now - Current epoch milliseconds.
|
|
@@ -41,6 +41,7 @@ import { POOLED_RUNNER_INFLIGHT_JOB_ID_LIMIT, boundedPooledRunnerInflightJobIds,
|
|
|
41
41
|
* @property {boolean} retiring - Whether this child is draining before retirement.
|
|
42
42
|
* @property {boolean} [started] - Whether the child completed its startup handshake.
|
|
43
43
|
* @property {boolean} [settling] - Whether failure handling already owns this child.
|
|
44
|
+
* @property {import("./types.js").PooledChildMemoryObservation} [lastMemoryObservation] - Latest memory observation received from this child.
|
|
44
45
|
* @property {number} [ipcDisconnectedAtMs] - Parent observation of IPC disconnect.
|
|
45
46
|
* @property {import("./types.js").PooledChildShutdownObservation} [shutdownObservation] - Child observation sent before teardown.
|
|
46
47
|
* @property {import("./types.js").PooledChildShutdownReason} [shutdownReason] - Exact parent-requested shutdown reason.
|
|
@@ -132,6 +133,38 @@ function isChildShutdownObservationMessage(message) {
|
|
|
132
133
|
&& isPooledChildShutdownSignal(record.signal)
|
|
133
134
|
}
|
|
134
135
|
|
|
136
|
+
/**
|
|
137
|
+
* Checks whether an IPC value is a pooled child's bounded memory observation.
|
|
138
|
+
* The pooled child's stdio is ignored by the worker fork, so this observation
|
|
139
|
+
* (periodic while jobs run, plus on demand) is how a memory problem in a
|
|
140
|
+
* running child names itself. The worker logs it (its stderr reaches the prod
|
|
141
|
+
* log) and forwards it to the optional `onPooledRunnerMemoryObservation` hook.
|
|
142
|
+
* @param {ReturnType<typeof JSON.parse>} message - IPC message.
|
|
143
|
+
* @returns {message is import("./types.js").PooledChildMemoryObservation} - Whether this is a valid memory observation.
|
|
144
|
+
*/
|
|
145
|
+
function isPooledChildMemoryObservationMessage(message) {
|
|
146
|
+
if (!message || typeof message !== "object") return false
|
|
147
|
+
const record = /** @type {Record<string, ReturnType<typeof JSON.parse>>} */ (message)
|
|
148
|
+
const heapStatistics = /** @type {Record<string, ReturnType<typeof JSON.parse>> | undefined} */ (record.heapStatistics)
|
|
149
|
+
const memoryUsage = /** @type {Record<string, ReturnType<typeof JSON.parse>> | undefined} */ (record.memoryUsage)
|
|
150
|
+
|
|
151
|
+
return record.type === "pooled-child-memory"
|
|
152
|
+
&& typeof record.childInstanceId === "string"
|
|
153
|
+
&& Number.isInteger(record.childPid)
|
|
154
|
+
&& typeof record.rssBytes === "number"
|
|
155
|
+
&& Number.isFinite(record.rssBytes)
|
|
156
|
+
&& Number.isInteger(record.jobCount)
|
|
157
|
+
&& record.jobCount >= 0
|
|
158
|
+
&& Array.isArray(record.activeJobIds)
|
|
159
|
+
&& record.activeJobIds.every((jobId) => typeof jobId === "string")
|
|
160
|
+
&& Number.isInteger(record.activeJobIdsTruncatedCount)
|
|
161
|
+
&& record.activeJobIdsTruncatedCount >= 0
|
|
162
|
+
&& typeof record.observedAtMs === "number"
|
|
163
|
+
&& Number.isFinite(record.observedAtMs)
|
|
164
|
+
&& heapStatistics !== undefined && typeof heapStatistics === "object"
|
|
165
|
+
&& memoryUsage !== undefined && typeof memoryUsage === "object"
|
|
166
|
+
}
|
|
167
|
+
|
|
135
168
|
/**
|
|
136
169
|
* Normalizes a candidate pooled-runner resource limit.
|
|
137
170
|
* @param {number | undefined} value - Candidate positive number.
|
|
@@ -166,8 +199,9 @@ export default class BackgroundJobsWorker {
|
|
|
166
199
|
* @param {() => void | Promise<void>} [args.onStopped] - Lifecycle hook invoked after the worker finishes stopping.
|
|
167
200
|
* @param {() => void} [args.onGenerationAccepted] - Explicit generation-acceptance observation hook.
|
|
168
201
|
* @param {() => void} [args.onRetireMessage] - Explicit retire-message observation hook.
|
|
202
|
+
* @param {(observation: import("./types.js").PooledChildMemoryObservation) => void | Promise<void>} [args.onPooledRunnerMemoryObservation] - Explicit pooled-child memory observation hook. Every validated observation (periodic while a child has in-flight jobs, plus on demand) is forwarded here, in addition to the worker's compact stderr log line, so an application can route memory diagnostics (e.g. to a bug reporter) without parsing logs.
|
|
169
203
|
*/
|
|
170
|
-
constructor({configuration, host, port, generationId, workerInstanceId, maxConcurrentForkedJobs, maxConcurrentInlineJobs, pooledRunnerCount, pooledRunnerConcurrency, pooledRunnerMaxJobs, pooledRunnerMaxRssBytes, pooledRunnerMaxLifetimeMs, forkedChildSigkillGraceMs, heartbeatIntervalMs, generationHandshakeTimeoutMs = DEFAULT_GENERATION_HANDSHAKE_TIMEOUT_MS, reconnectDelayMs = 1000, jobTimeoutMs, closeDatabaseConnectionsOnStop = true, onStopped, onGenerationAccepted, onRetireMessage} = {}) {
|
|
204
|
+
constructor({configuration, host, port, generationId, workerInstanceId, maxConcurrentForkedJobs, maxConcurrentInlineJobs, pooledRunnerCount, pooledRunnerConcurrency, pooledRunnerMaxJobs, pooledRunnerMaxRssBytes, pooledRunnerMaxLifetimeMs, forkedChildSigkillGraceMs, heartbeatIntervalMs, generationHandshakeTimeoutMs = DEFAULT_GENERATION_HANDSHAKE_TIMEOUT_MS, reconnectDelayMs = 1000, jobTimeoutMs, closeDatabaseConnectionsOnStop = true, onStopped, onGenerationAccepted, onRetireMessage, onPooledRunnerMemoryObservation} = {}) {
|
|
171
205
|
/**
|
|
172
206
|
* Narrows the runtime value to the documented type.
|
|
173
207
|
* @type {Promise<import("../configuration.js").default>} */
|
|
@@ -186,6 +220,7 @@ export default class BackgroundJobsWorker {
|
|
|
186
220
|
this.onStopped = onStopped
|
|
187
221
|
this.onGenerationAccepted = onGenerationAccepted
|
|
188
222
|
this.onRetireMessage = onRetireMessage
|
|
223
|
+
this.onPooledRunnerMemoryObservation = onPooledRunnerMemoryObservation
|
|
189
224
|
/**
|
|
190
225
|
* Constructor override for the inline-job concurrency cap. When unset
|
|
191
226
|
* the cap is read from `configuration.getBackgroundJobsConfig()` in
|
|
@@ -322,6 +357,28 @@ export default class BackgroundJobsWorker {
|
|
|
322
357
|
// Monotonic dispatch counter for round-robin child selection: each dispatch stamps
|
|
323
358
|
// the chosen child, and selection prefers the child dispatched least recently.
|
|
324
359
|
this._pooledDispatchSeq = 0
|
|
360
|
+
// Waiters blocked in _runPooledJob because the pool is at its hard cap: a job may
|
|
361
|
+
// not spawn a child while total live children (working + draining) is at the cap.
|
|
362
|
+
/** @type {Set<() => void>} */
|
|
363
|
+
this._pooledSlotWaiters = new Set()
|
|
364
|
+
/** @type {ReturnType<typeof setInterval> | undefined} - Safety poll that re-checks the slot condition. */
|
|
365
|
+
this._pooledSlotWaitTimer = undefined
|
|
366
|
+
}
|
|
367
|
+
|
|
368
|
+
/** Starts the slot-waiter safety poll if it is not already running. */
|
|
369
|
+
_startPooledSlotWaitPoll() {
|
|
370
|
+
if (this._pooledSlotWaitTimer) return
|
|
371
|
+
|
|
372
|
+
this._pooledSlotWaitTimer = setInterval(() => this._wakePooledSlotWaiters(), 50)
|
|
373
|
+
this._pooledSlotWaitTimer.unref()
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
/** Stops the slot-waiter safety poll once no waiter is registered. */
|
|
377
|
+
_stopPooledSlotWaitPollIfIdle() {
|
|
378
|
+
if (this._pooledSlotWaiters.size > 0 || !this._pooledSlotWaitTimer) return
|
|
379
|
+
|
|
380
|
+
clearInterval(this._pooledSlotWaitTimer)
|
|
381
|
+
this._pooledSlotWaitTimer = undefined
|
|
325
382
|
}
|
|
326
383
|
|
|
327
384
|
/**
|
|
@@ -1079,27 +1136,72 @@ export default class BackgroundJobsWorker {
|
|
|
1079
1136
|
}
|
|
1080
1137
|
}
|
|
1081
1138
|
|
|
1139
|
+
/**
|
|
1140
|
+
* Hard cap on total live pooled children (working + draining): children the
|
|
1141
|
+
* pool may still spawn. Counting the whole live set — draining children
|
|
1142
|
+
* included — is what bounds pool memory: a draining child still holds its
|
|
1143
|
+
* RSS until its last in-flight job finishes, so it must occupy a cap slot.
|
|
1144
|
+
* @returns {number} - Number of children the pool may still spawn.
|
|
1145
|
+
*/
|
|
1146
|
+
_spawnablePooledChildren() {
|
|
1147
|
+
return Math.max(0, this.pooledRunnerCount - this.pooledChildren.size)
|
|
1148
|
+
}
|
|
1149
|
+
|
|
1150
|
+
/**
|
|
1151
|
+
* Resolves once a non-retiring pooled child has a free concurrency slot or the
|
|
1152
|
+
* pool may spawn a new one. Pooled jobs admitted while the pool is at its
|
|
1153
|
+
* hard cap wait here instead of spawning an over-capacity child; the wake
|
|
1154
|
+
* points are the only capacity-freeing transitions (a job outcome and a child
|
|
1155
|
+
* exit), so no polling is needed.
|
|
1156
|
+
* @returns {Promise<void>} - Resolves when a slot is available.
|
|
1157
|
+
*/
|
|
1158
|
+
_waitPooledSlot() {
|
|
1159
|
+
this._startPooledSlotWaitPoll()
|
|
1160
|
+
|
|
1161
|
+
return new Promise((resolve) => {
|
|
1162
|
+
const waiter = () => {
|
|
1163
|
+
this._pooledSlotWaiters.delete(waiter)
|
|
1164
|
+
this._stopPooledSlotWaitPollIfIdle()
|
|
1165
|
+
resolve()
|
|
1166
|
+
}
|
|
1167
|
+
|
|
1168
|
+
this._pooledSlotWaiters.add(waiter)
|
|
1169
|
+
})
|
|
1170
|
+
}
|
|
1171
|
+
|
|
1172
|
+
/**
|
|
1173
|
+
* Resolves every registered waiter; waiters re-check the slot condition
|
|
1174
|
+
* themselves and only proceed when it holds.
|
|
1175
|
+
* @returns {void}
|
|
1176
|
+
*/
|
|
1177
|
+
_wakePooledSlotWaiters() {
|
|
1178
|
+
if (this._pooledSlotWaiters.size === 0) return
|
|
1179
|
+
|
|
1180
|
+
for (const resolve of [...this._pooledSlotWaiters]) resolve()
|
|
1181
|
+
}
|
|
1182
|
+
|
|
1082
1183
|
/**
|
|
1083
1184
|
* Free pooled slots across the pool: open slots in non-retiring children plus
|
|
1084
|
-
* the slots we could add by spawning more children up to
|
|
1085
|
-
* Retiring children (draining before replacement) never
|
|
1185
|
+
* the slots we could add by spawning more children up to the hard cap on total
|
|
1186
|
+
* live children. Retiring children (draining before replacement) never
|
|
1187
|
+
* contribute capacity, and they count against the cap: while one is still
|
|
1188
|
+
* draining, no replacement is advertised (or spawned) — the pool advertises
|
|
1189
|
+
* exactly what it can serve instead of phantom capacity.
|
|
1086
1190
|
* @returns {number} - Number of pooled jobs the worker can accept right now.
|
|
1087
1191
|
*/
|
|
1088
1192
|
_availablePooledSlots() {
|
|
1089
1193
|
let openInExisting = 0
|
|
1090
|
-
let nonRetiringChildren = 0
|
|
1091
1194
|
let queuedReservations = 0
|
|
1092
1195
|
|
|
1093
1196
|
for (const child of this.pooledChildren) {
|
|
1094
1197
|
const state = this.pooledChildStates.get(child)
|
|
1095
1198
|
if (!state || state.retiring) continue
|
|
1096
|
-
nonRetiringChildren += 1
|
|
1097
1199
|
openInExisting += this.pooledRunnerConcurrency - state.inflight.size
|
|
1098
1200
|
}
|
|
1099
1201
|
|
|
1100
1202
|
for (const queue of this.pooledJobQueues.values()) queuedReservations += queue.length
|
|
1101
1203
|
|
|
1102
|
-
const spawnableChildren = Math.max(0, this.pooledRunnerCount -
|
|
1204
|
+
const spawnableChildren = Math.max(0, this.pooledRunnerCount - this.pooledChildren.size)
|
|
1103
1205
|
|
|
1104
1206
|
return Math.max(0, openInExisting + spawnableChildren * this.pooledRunnerConcurrency - queuedReservations)
|
|
1105
1207
|
}
|
|
@@ -1109,11 +1211,32 @@ export default class BackgroundJobsWorker {
|
|
|
1109
1211
|
* new child when every non-retiring child is full and the pool is below
|
|
1110
1212
|
* `pooledRunnerCount`. Each child runs up to `pooledRunnerConcurrency` jobs at
|
|
1111
1213
|
* once on its own event loop.
|
|
1214
|
+
*
|
|
1215
|
+
* When the pool is already at its hard cap (total live children, draining
|
|
1216
|
+
* included), the job waits for a slot instead of spawning: that is what keeps
|
|
1217
|
+
* the live-child count — and therefore the pool's total RSS — bounded. The
|
|
1218
|
+
* wait resolves on the only two capacity-freeing transitions (a job outcome,
|
|
1219
|
+
* a child exit); a safety poll covers anything missed.
|
|
1112
1220
|
* @param {import("./types.js").BackgroundJobPayload & {id: string}} payload - Job payload.
|
|
1113
1221
|
* @returns {Promise<void>} - Resolves after the durable report.
|
|
1114
1222
|
*/
|
|
1115
|
-
_runPooledJob(payload) {
|
|
1116
|
-
|
|
1223
|
+
async _runPooledJob(payload) {
|
|
1224
|
+
// At the hard cap (no free slot, no spawnable child) the job waits for a
|
|
1225
|
+
// slot instead of spawning an over-capacity child — that is what bounds
|
|
1226
|
+
// the live-child count and the pool's total RSS.
|
|
1227
|
+
let child = this._selectPooledChild()
|
|
1228
|
+
while (!child) {
|
|
1229
|
+
if (this._spawnablePooledChildren() === 0) {
|
|
1230
|
+
// Shutdown: main no longer dispatches and no slot will ever free —
|
|
1231
|
+
// stop waiting so the tracked job can settle and the drain completes.
|
|
1232
|
+
if (this.shouldStop) return
|
|
1233
|
+
await this._waitPooledSlot()
|
|
1234
|
+
child = this._selectPooledChild()
|
|
1235
|
+
continue
|
|
1236
|
+
}
|
|
1237
|
+
child = this._selectPooledChild() || this._createPooledChild()
|
|
1238
|
+
}
|
|
1239
|
+
|
|
1117
1240
|
const state = this.pooledChildStates.get(child)
|
|
1118
1241
|
if (!state) throw new Error("Pooled runner state missing")
|
|
1119
1242
|
|
|
@@ -1232,12 +1355,16 @@ export default class BackgroundJobsWorker {
|
|
|
1232
1355
|
}
|
|
1233
1356
|
|
|
1234
1357
|
/**
|
|
1235
|
-
* Creates a reusable pooled child
|
|
1236
|
-
*
|
|
1358
|
+
* Creates a reusable pooled child, enforcing the hard cap on total live
|
|
1359
|
+
* children (working + draining). Returns undefined when the cap is already
|
|
1360
|
+
* met — the only way a new child may exist is a slot being open, so the
|
|
1361
|
+
* caller re-checks and waits again.
|
|
1362
|
+
* @returns {import("node:child_process").ChildProcess | undefined} - The new child, or undefined when the pool is at its cap.
|
|
1237
1363
|
*/
|
|
1238
1364
|
_createPooledChild() {
|
|
1239
1365
|
const configuration = this.configuration
|
|
1240
1366
|
if (!configuration) throw new Error("Background jobs worker configuration not initialized")
|
|
1367
|
+
if (this.pooledChildren.size >= this.pooledRunnerCount) return undefined
|
|
1241
1368
|
const child = fork(POOLED_RUNNER_ENTRY_PATH, [], {
|
|
1242
1369
|
cwd: configuration.getDirectory(), execArgv: [], stdio: ["ignore", "ignore", "ignore", "ipc"],
|
|
1243
1370
|
env: Object.assign({}, process.env, this._childBackgroundJobsEnvironment())
|
|
@@ -1301,6 +1428,10 @@ export default class BackgroundJobsWorker {
|
|
|
1301
1428
|
this._reportChildAccepted(message)
|
|
1302
1429
|
return
|
|
1303
1430
|
}
|
|
1431
|
+
if (isPooledChildMemoryObservationMessage(message)) {
|
|
1432
|
+
this._handlePooledChildMemoryObservation({child, message})
|
|
1433
|
+
return
|
|
1434
|
+
}
|
|
1304
1435
|
if (record.type !== "job-outcome" || !state || state.settling || typeof record.jobId !== "string") return
|
|
1305
1436
|
state.started = true
|
|
1306
1437
|
const entry = state.inflight.get(record.jobId)
|
|
@@ -1332,6 +1463,9 @@ export default class BackgroundJobsWorker {
|
|
|
1332
1463
|
this._beginRetirePooledChild(child)
|
|
1333
1464
|
}
|
|
1334
1465
|
this._terminateIfDrained(child)
|
|
1466
|
+
// A job outcome frees a concurrency slot and may drain a retiring child —
|
|
1467
|
+
// the only two transitions that free capacity for waiters at the hard cap.
|
|
1468
|
+
this._wakePooledSlotWaiters()
|
|
1335
1469
|
}
|
|
1336
1470
|
|
|
1337
1471
|
/**
|
|
@@ -1362,11 +1496,85 @@ export default class BackgroundJobsWorker {
|
|
|
1362
1496
|
}
|
|
1363
1497
|
|
|
1364
1498
|
/**
|
|
1365
|
-
*
|
|
1366
|
-
*
|
|
1367
|
-
*
|
|
1368
|
-
*
|
|
1369
|
-
*
|
|
1499
|
+
* Handles a pooled child's bounded memory observation. The pooled child's
|
|
1500
|
+
* stdio is ignored by the worker fork, so this IPC observation is how a
|
|
1501
|
+
* memory problem in a running child names itself. The worker (a) records the
|
|
1502
|
+
* latest observation on the child's state for later correlation, (b) logs one
|
|
1503
|
+
* compact line to its own stderr (which reaches the prod log, unlike the
|
|
1504
|
+
* child's ignored stdio), and (c) forwards the full observation to the
|
|
1505
|
+
* optional `onPooledRunnerMemoryObservation` hook so an application can route
|
|
1506
|
+
* it (e.g. to a bug reporter) without parsing logs. The heap-stat breakdown
|
|
1507
|
+
* distinguishes V8-heap growth from external/array-buffer (native) growth. A
|
|
1508
|
+
* hook failure is swallowed — diagnostics must never take down the worker or
|
|
1509
|
+
* fail the jobs running on that child.
|
|
1510
|
+
* @param {object} args - Message details.
|
|
1511
|
+
* @param {import("node:child_process").ChildProcess} args.child - Pooled child.
|
|
1512
|
+
* @param {import("./types.js").PooledChildMemoryObservation} args.message - Validated memory observation.
|
|
1513
|
+
* @returns {void}
|
|
1514
|
+
*/
|
|
1515
|
+
_handlePooledChildMemoryObservation({child, message}) {
|
|
1516
|
+
const state = this.pooledChildStates.get(child)
|
|
1517
|
+
if (state) state.lastMemoryObservation = message
|
|
1518
|
+
|
|
1519
|
+
const heap = message.heapStatistics
|
|
1520
|
+
console.error(
|
|
1521
|
+
JSON.stringify({
|
|
1522
|
+
event: "pooled-child-memory",
|
|
1523
|
+
childInstanceId: state?.childInstanceId ?? message.childInstanceId,
|
|
1524
|
+
childPid: message.childPid,
|
|
1525
|
+
childUptimeS: Math.round(message.childUptimeMs / 1000),
|
|
1526
|
+
rssMb: Math.round(message.rssBytes / (1024 * 1024)),
|
|
1527
|
+
heapUsedMb: Math.round(heap.used_heap_size / (1024 * 1024)),
|
|
1528
|
+
heapTotalMb: Math.round(heap.total_heap_size / (1024 * 1024)),
|
|
1529
|
+
heapLimitMb: Math.round(heap.heap_size_limit / (1024 * 1024)),
|
|
1530
|
+
externalMb: Math.round(message.memoryUsage.external / (1024 * 1024)),
|
|
1531
|
+
arrayBuffersMb: Math.round(message.memoryUsage.arrayBuffers / (1024 * 1024)),
|
|
1532
|
+
jobs: message.jobCount,
|
|
1533
|
+
activeJobIds: message.activeJobIds
|
|
1534
|
+
})
|
|
1535
|
+
)
|
|
1536
|
+
|
|
1537
|
+
if (this.onPooledRunnerMemoryObservation) {
|
|
1538
|
+
try {
|
|
1539
|
+
const result = this.onPooledRunnerMemoryObservation(message)
|
|
1540
|
+
if (result && typeof result.catch === "function") result.catch((error) => {
|
|
1541
|
+
console.error("Pooled runner memory observation hook failed:", error)
|
|
1542
|
+
})
|
|
1543
|
+
} catch (error) {
|
|
1544
|
+
console.error("Pooled runner memory observation hook failed:", error)
|
|
1545
|
+
}
|
|
1546
|
+
}
|
|
1547
|
+
}
|
|
1548
|
+
|
|
1549
|
+
/**
|
|
1550
|
+
* Requests an immediate memory observation from one pooled child. The child
|
|
1551
|
+
* replies over IPC with its current snapshot (the same shape as the periodic
|
|
1552
|
+
* sampler), which the worker records, logs, and forwards to the
|
|
1553
|
+
* `onPooledRunnerMemoryObservation` hook. Use this to pull a snapshot on
|
|
1554
|
+
* suspicion (e.g. after an OOM report) without waiting for the next periodic
|
|
1555
|
+
* sample. A no-op when the child is gone or its IPC channel is closed.
|
|
1556
|
+
* @param {import("node:child_process").ChildProcess} child - Pooled child to sample.
|
|
1557
|
+
* @returns {void}
|
|
1558
|
+
*/
|
|
1559
|
+
requestPooledChildMemoryObservation(child) {
|
|
1560
|
+
if (!child.connected) return
|
|
1561
|
+
|
|
1562
|
+
try {
|
|
1563
|
+
child.send({type: "memory-observation-request"})
|
|
1564
|
+
} catch {
|
|
1565
|
+
// The IPC channel is already gone; the disconnect/exit handler owns teardown.
|
|
1566
|
+
}
|
|
1567
|
+
}
|
|
1568
|
+
|
|
1569
|
+
/**
|
|
1570
|
+
* Marks a pooled child for retirement and — when the pool is below its hard
|
|
1571
|
+
* cap — eagerly spawns a single replacement (1-for-1) so its capacity is
|
|
1572
|
+
* restored immediately without waiting for it to finish draining. The
|
|
1573
|
+
* replacement spawn is gated by the cap (the retiring child still counts as
|
|
1574
|
+
* live until it exits), so a full pool simply defers the replacement to the
|
|
1575
|
+
* retiring child's drain instead of spawning over capacity. The retiring
|
|
1576
|
+
* child stops receiving new jobs and is terminated only once its in-flight
|
|
1577
|
+
* set drains, so a long-running job (e.g. a build) is never cut off.
|
|
1370
1578
|
* @param {import("node:child_process").ChildProcess} child - Child to retire.
|
|
1371
1579
|
* @returns {void}
|
|
1372
1580
|
*/
|
|
@@ -1376,7 +1584,9 @@ export default class BackgroundJobsWorker {
|
|
|
1376
1584
|
|
|
1377
1585
|
state.retiring = true
|
|
1378
1586
|
// Best-effort pre-warm: skip when stopping (no new work) or before the
|
|
1379
|
-
// worker is initialized (no configuration to fork a child from).
|
|
1587
|
+
// worker is initialized (no configuration to fork a child from). The cap
|
|
1588
|
+
// inside _createPooledChild refuses the spawn while the pool is full, in
|
|
1589
|
+
// which case the replacement is deferred to the drain path.
|
|
1380
1590
|
if (!this.shouldStop && this.configuration) this._createPooledChild()
|
|
1381
1591
|
}
|
|
1382
1592
|
|
|
@@ -1394,6 +1604,9 @@ export default class BackgroundJobsWorker {
|
|
|
1394
1604
|
|
|
1395
1605
|
/**
|
|
1396
1606
|
* Retires a drained pooled child (removes it from tracking, then SIGTERMs it).
|
|
1607
|
+
* Because the hard cap counts live children, the exit of this child frees a
|
|
1608
|
+
* slot: any deferred replacement (the pool was full when the child retired)
|
|
1609
|
+
* is spawned now, and capacity is re-advertised so main can dispatch into it.
|
|
1397
1610
|
* @param {import("node:child_process").ChildProcess} child - Child process to retire.
|
|
1398
1611
|
* @returns {void}
|
|
1399
1612
|
*/
|
|
@@ -1503,6 +1716,9 @@ export default class BackgroundJobsWorker {
|
|
|
1503
1716
|
}
|
|
1504
1717
|
this.pooledChildren.delete(child)
|
|
1505
1718
|
this.inflightProcessChildren.delete(child)
|
|
1719
|
+
// Child exit frees a hard-cap slot even while its in-flight set is still
|
|
1720
|
+
// being reported — wake waiters now; their reports settle independently.
|
|
1721
|
+
this._wakePooledSlotWaiters()
|
|
1506
1722
|
|
|
1507
1723
|
const entries = state ? [...state.inflight.values()] : []
|
|
1508
1724
|
const runnerFailure = state
|