queen-mq 1.0.5 → 1.0.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -324,6 +324,21 @@ await queen.flushAllBuffers()
324
324
  // Result: 10x-100x faster than individual pushes
325
325
  ```
326
326
 
327
+ The buffer is **bounded and lossless**, and both properties are why `push()` must
328
+ be awaited:
329
+
330
+ | Option | Default | Meaning |
331
+ | --- | --- | --- |
332
+ | `messageCount` | `100` | Flush once this many messages are waiting |
333
+ | `timeMillis` | `1000` | Or this long after the first message arrives |
334
+ | `maxSize` | `4 x messageCount` | Backpressure bound: past this many buffered messages, `push()` WAITS for the flusher instead of growing the heap. There is no unbounded setting |
335
+ | `retryDelayMillis` | `250` | Delay before retrying a batch whose POST failed. Failed batches go back to the front of the buffer, in order, and are retried — never dropped |
336
+
337
+ A producer that outruns the flush pipeline is therefore paced down to the drain
338
+ rate, and a broker outage shows up as slow pushes with bounded memory rather
339
+ than as messages that quietly disappeared. `close()` flushes with a 30 second
340
+ deadline and logs how many messages were left unsent if it expires.
341
+
327
342
  ### Dead Letter Queue
328
343
 
329
344
  ```javascript
@@ -17,6 +17,12 @@ import { CLIENT_DEFAULTS } from './utils/defaults.js'
17
17
  import { validateUrl, validateUrls } from './utils/validation.js'
18
18
  import * as logger from './utils/logger.js'
19
19
 
20
+ // How long close() keeps retrying a push batch the broker will not take before
21
+ // it gives up, logs how many messages were never sent, and lets the process
22
+ // exit. Matches CLIENT_DEFAULTS.timeoutMillis and the usual 30s SIGTERM grace:
23
+ // long enough to ride out a broker restart, short enough that shutdown ends.
24
+ const CLOSE_FLUSH_DEADLINE_MILLIS = 30000
25
+
20
26
  // Both /api/v1/ack and /api/v1/ack/batch respond with a top-level JSON array,
21
27
  // one item per acknowledgment in request order:
22
28
  // [{index, transactionId, success, error, queueName, partitionName, leaseReleased, dlq}]
@@ -596,9 +602,13 @@ export class Queen {
596
602
  async close() {
597
603
  logger.log('Queen.close', 'Starting shutdown')
598
604
 
599
- // Flush all buffers
605
+ // Flush all buffers, with a deadline. The flusher retries a failed batch
606
+ // forever rather than dropping it, which is right while the process is
607
+ // running and wrong on the way out: a SIGTERM grace period is finite, so
608
+ // shutdown stops retrying after CLOSE_FLUSH_DEADLINE_MILLIS and reports
609
+ // what is left instead of hanging until the runtime is killed.
600
610
  try {
601
- await this.#bufferManager.flushAllBuffers()
611
+ await this.#bufferManager.flushAllBuffers({ deadlineMillis: CLOSE_FLUSH_DEADLINE_MILLIS })
602
612
  logger.log('Queen.close', 'All buffers flushed')
603
613
  } catch (error) {
604
614
  logger.error('Queen.close', { error: error.message, phase: 'buffer-flush' })
@@ -1,176 +1,220 @@
1
1
  /**
2
- * Buffer manager for client-side message buffering across queues
2
+ * Buffer manager for client-side message buffering across queues.
3
+ *
4
+ * One MessageBuffer per `queue/partition` address (the granularity the broker
5
+ * fuses writes on), and exactly ONE drain loop per buffer. The drain is the
6
+ * only thing that sends: it takes `messageCount`-sized batches off the front,
7
+ * POSTs them, and wakes producers parked on the buffer's maxSize bound after
8
+ * each batch that is definitively gone.
9
+ *
10
+ * A batch whose POST fails goes straight back to the front of the buffer, in
11
+ * order, and is retried after `retryDelayMillis` -- indefinitely, until it
12
+ * lands or the buffer is stopped. That is the half of the 2026-08-20 fix that
13
+ * removes loss on flush error; MessageBuffer's docs carry the other half
14
+ * (blocking backpressure) and the measurements behind both.
15
+ *
16
+ * Deadlines: an explicit flush (`flushBuffer`, `flushAllBuffers`) may pass
17
+ * `deadlineMillis` to bound how long it is willing to keep retrying, because
18
+ * "retry forever" is right for a background flusher and wrong for a shutdown
19
+ * path that has a SIGTERM grace period to respect. When the deadline expires
20
+ * the messages are still in the buffer -- the error says how many -- so the
21
+ * failure is loud rather than silent.
3
22
  */
4
23
 
5
24
  import { MessageBuffer } from './MessageBuffer.js'
6
- import { BUFFER_DEFAULTS } from '../utils/defaults.js'
7
25
  import * as logger from '../utils/logger.js'
8
26
 
9
27
  export class BufferManager {
10
28
  #httpClient
11
29
  #buffers = new Map() // queueAddress -> MessageBuffer
12
- #pendingFlushes = new Set() // Track in-flight flush promises
30
+ #drains = new Map() // queueAddress -> { promise, ctl } for the in-flight drain
13
31
  #flushCount = 0
32
+ #stopped = false
14
33
 
15
34
  constructor(httpClient) {
16
35
  this.#httpClient = httpClient
17
36
  }
18
37
 
19
- addMessage(queueAddress, formattedMessage, bufferOptions) {
20
- const options = { ...BUFFER_DEFAULTS, ...bufferOptions }
38
+ /**
39
+ * Buffer one message, waiting for room if the buffer is at its bound.
40
+ *
41
+ * Returns a promise: the add path is where backpressure is applied, so
42
+ * callers MUST await it. PushBuilder does; anything that forgets would be
43
+ * back to the unbounded behaviour this replaced.
44
+ *
45
+ * @param {string} queueAddress
46
+ * @param {object} formattedMessage
47
+ * @param {object} bufferOptions
48
+ * @param {{ signal?: AbortSignal }} [opts]
49
+ */
50
+ async addMessage(queueAddress, formattedMessage, bufferOptions, { signal } = {}) {
51
+ // A push after cleanup() would otherwise create a fresh buffer that nothing
52
+ // will ever flush -- messages accepted into a client that is already shut
53
+ // down, which is the same false success the bound exists to remove.
54
+ if (this.#stopped) {
55
+ throw new Error(`Queen client is closed: message not buffered for ${queueAddress}`)
56
+ }
21
57
 
22
58
  if (!this.#buffers.has(queueAddress)) {
23
- logger.log('BufferManager.createBuffer', { queueAddress, options })
24
- this.#buffers.set(queueAddress, new MessageBuffer(
25
- queueAddress,
26
- options,
27
- (addr) => this.#flushBuffer(addr)
28
- ))
59
+ // The raw options go to the buffer, which fills in the defaults itself:
60
+ // maxSize is derived from the messageCount this caller asked for, and
61
+ // merging defaults here first would hide the difference between "not set"
62
+ // and "set to the default".
63
+ const created = new MessageBuffer(queueAddress, bufferOptions, (addr) => { this.#startDrain(addr) })
64
+ logger.log('BufferManager.createBuffer', { queueAddress, options: created.options })
65
+ this.#buffers.set(queueAddress, created)
29
66
  }
30
67
 
31
68
  const buffer = this.#buffers.get(queueAddress)
32
- buffer.add(formattedMessage)
69
+ await buffer.add(formattedMessage, { signal })
33
70
  logger.log('BufferManager.addMessage', { queueAddress, messageCount: buffer.messageCount })
34
71
  }
35
72
 
36
- async #flushBuffer(queueAddress) {
37
- const buffer = this.#buffers.get(queueAddress)
38
- if (!buffer || buffer.messageCount === 0) {
39
- logger.debug('BufferManager.flushBuffer', { queueAddress, status: 'empty' })
40
- return
73
+ /**
74
+ * Start the drain loop for an address, or join the one already running.
75
+ *
76
+ * Joining rather than starting a second loop is what keeps batches in order:
77
+ * two concurrent senders on the same partition would interleave their POSTs.
78
+ * Returns null when there is nothing to send.
79
+ *
80
+ * @param {string} queueAddress
81
+ * @param {number|null} deadlineMillis - how long a caller is willing to keep
82
+ * retrying a failing batch; null (the default, and what background flushes
83
+ * use) means "until it lands or the buffer stops".
84
+ */
85
+ #startDrain(queueAddress, deadlineMillis = null) {
86
+ const deadline = deadlineMillis === null || deadlineMillis === undefined
87
+ ? Number.POSITIVE_INFINITY
88
+ : Date.now() + deadlineMillis
89
+
90
+ const running = this.#drains.get(queueAddress)
91
+ if (running) {
92
+ // A caller with a deadline joining a background drain tightens it: the
93
+ // shortest patience wins, otherwise a shutdown could be held open by a
94
+ // retry loop that was started with none.
95
+ running.ctl.deadline = Math.min(running.ctl.deadline, deadline)
96
+ return running.promise
41
97
  }
42
98
 
43
- logger.log('BufferManager.flushBuffer', { queueAddress, messageCount: buffer.messageCount })
44
- buffer.setFlushing(true)
45
-
46
- // Create a promise for this flush and track it
47
- const flushPromise = (async () => {
48
- try {
49
- const messages = buffer.extractMessages()
50
-
51
- logger.debug('BufferManager.flushBuffer', { queueAddress, extracted: messages.length })
52
-
53
- if (messages.length === 0) return
54
-
55
- // Send to server
56
- const result = await this.#httpClient.post('/api/v1/push', { items: messages })
57
- logger.debug('BufferManager.flushBuffer', { queueAddress, serverResponse: result ? `${result.length || 'N/A'} items` : 'null' })
58
-
59
- this.#flushCount++
60
- logger.log('BufferManager.flushBuffer', { queueAddress, status: 'success', messagesSent: messages.length })
61
-
62
- // Remove empty buffer
63
- this.#buffers.delete(queueAddress)
64
-
65
- } catch (error) {
66
- logger.error('BufferManager.flushBuffer', { queueAddress, error: error.message })
67
- buffer.setFlushing(false)
68
- throw error
69
- } finally {
70
- // Remove from pending flushes
71
- this.#pendingFlushes.delete(flushPromise)
72
- }
73
- })()
74
-
75
- // Track this flush
76
- this.#pendingFlushes.add(flushPromise)
77
-
78
- return flushPromise
79
- }
80
-
81
- async #flushBufferBatch(queueAddress, batchSize) {
82
99
  const buffer = this.#buffers.get(queueAddress)
83
- if (!buffer || buffer.messageCount === 0) {
84
- return
85
- }
86
-
87
- buffer.setFlushing(true)
88
-
89
- // Create a promise for this flush and track it
90
- const flushPromise = (async () => {
91
- try {
92
- const messages = buffer.extractMessages(batchSize)
93
-
94
- logger.debug('BufferManager.flushBufferBatch', { queueAddress, extracted: messages.length })
95
-
96
- if (messages.length === 0) return
97
-
98
- // Send to server
99
- const result = await this.#httpClient.post('/api/v1/push', { items: messages })
100
- logger.debug('BufferManager.flushBufferBatch', { queueAddress, serverResponse: result ? `${result.length || 'N/A'} items` : 'null' })
101
-
102
- this.#flushCount++
100
+ if (!buffer || !buffer.beginFlush()) return null
101
+
102
+ const ctl = { deadline }
103
+ const promise = this.#drain(queueAddress, buffer, ctl)
104
+ this.#drains.set(queueAddress, { promise, ctl })
105
+ // The count-threshold and timer triggers do not await this promise, so give
106
+ // it a handler of its own: a deadline tightened by a concurrent explicit
107
+ // flush would otherwise surface as an unhandled rejection.
108
+ promise.catch(() => {})
109
+ return promise
110
+ }
103
111
 
104
- // Remove empty buffer if no more messages
105
- if (buffer.messageCount === 0) {
106
- this.#buffers.delete(queueAddress)
107
- } else {
108
- buffer.setFlushing(false)
112
+ async #drain(queueAddress, buffer, ctl) {
113
+ const { messageCount, retryDelayMillis } = buffer.options
114
+
115
+ try {
116
+ while (buffer.messageCount > 0 && !buffer.isStopped) {
117
+ const batch = buffer.takeBatch(messageCount)
118
+ if (batch.length === 0) break
119
+
120
+ try {
121
+ const result = await this.#httpClient.post('/api/v1/push', { items: batch })
122
+ logger.debug('BufferManager.drain', { queueAddress, sent: batch.length, serverResponse: result ? `${result.length || 'N/A'} items` : 'null' })
123
+
124
+ this.#flushCount++
125
+ // Capacity freed. Woken here rather than at takeBatch, because a
126
+ // batch that fails to send goes straight back: waking on extraction
127
+ // would let producers refill against room that never freed.
128
+ buffer.wakeWaiters()
129
+ } catch (error) {
130
+ // NOT dropped. The batch goes back at the front, in order, and this
131
+ // loop retries it. Before 2026-08-20 this branch logged and moved on,
132
+ // losing up to messageCount messages per failed POST.
133
+ buffer.restoreBatch(batch)
134
+ logger.error('BufferManager.drain', { queueAddress, error: error.message, requeued: batch.length })
135
+
136
+ const remaining = ctl.deadline - Date.now()
137
+ if (remaining <= 0) {
138
+ error.queenUnflushedCount = buffer.messageCount
139
+ error.message = `${error.message} (${buffer.messageCount} message(s) still buffered for ${queueAddress}, not sent)`
140
+ throw error
141
+ }
142
+
143
+ await buffer.sleepUnlessStopped(Math.min(retryDelayMillis, remaining))
109
144
  }
110
-
111
- } catch (error) {
112
- logger.error('BufferManager.flushBufferBatch', { queueAddress, error: error.message })
113
- buffer.setFlushing(false)
114
- throw error
115
- } finally {
116
- // Remove from pending flushes
117
- this.#pendingFlushes.delete(flushPromise)
118
145
  }
119
- })()
120
-
121
- // Track this flush
122
- this.#pendingFlushes.add(flushPromise)
123
-
124
- return flushPromise
146
+ } finally {
147
+ buffer.endFlush()
148
+ this.#drains.delete(queueAddress)
149
+ // Drop the entry only when nothing can still be pointed at it: a parked
150
+ // add holds this exact object, and deleting it here would leave that add
151
+ // appending into an orphan no drain would ever visit.
152
+ if (buffer.messageCount === 0 && !buffer.hasParkedAdds && !buffer.isStopped) {
153
+ this.#buffers.delete(queueAddress)
154
+ }
155
+ buffer.wakeWaiters()
156
+ }
125
157
  }
126
158
 
127
- async flushBuffer(queueAddress) {
128
- logger.log('BufferManager.flushBuffer', { queueAddress, activeBuffers: this.#buffers.size, pendingFlushes: this.#pendingFlushes.size })
129
-
159
+ /**
160
+ * Send everything buffered for one address.
161
+ *
162
+ * @param {string} queueAddress
163
+ * @param {{ deadlineMillis?: number }} [opts] - stop retrying a failing batch
164
+ * after this long and throw (the messages stay in the buffer). Omit to
165
+ * retry until the batch lands.
166
+ */
167
+ async flushBuffer(queueAddress, { deadlineMillis = null } = {}) {
168
+ logger.log('BufferManager.flushBuffer', { queueAddress, activeBuffers: this.#buffers.size, pendingFlushes: this.#drains.size })
169
+
130
170
  const buffer = this.#buffers.get(queueAddress)
131
171
  if (!buffer) {
132
172
  logger.debug('BufferManager.flushBuffer', { queueAddress, status: 'not-found' })
133
- await this.#waitForPendingFlushes()
173
+ await this.#waitForDrains()
134
174
  return
135
175
  }
136
176
 
137
- // Cancel timer to prevent time-based flush
177
+ // Cancel the timer to prevent a time-based flush racing this one.
138
178
  buffer.cancelTimer()
139
-
140
- // Get the batch size from buffer options
141
- const batchSize = buffer.options.messageCount
142
-
143
- // Flush all messages in batches
144
- while (buffer.messageCount > 0) {
145
- logger.debug('BufferManager.flushBuffer', { queueAddress, batchSize, remaining: buffer.messageCount })
146
- await this.#flushBufferBatch(queueAddress, batchSize)
147
- }
148
-
149
- // Wait for all pending flushes to complete
150
- await this.#waitForPendingFlushes()
151
-
179
+
180
+ const drain = this.#startDrain(queueAddress, deadlineMillis)
181
+ if (drain) await drain
182
+
152
183
  logger.debug('BufferManager.flushBuffer', { queueAddress, status: 'completed' })
153
184
  }
154
185
 
155
- async flushAllBuffers() {
156
- // Get all queue addresses that have buffers
157
- const queueAddresses = Array.from(this.#buffers.keys())
158
- logger.log('BufferManager.flushAllBuffers', { bufferCount: queueAddresses.length, pendingFlushes: this.#pendingFlushes.size })
159
-
160
- // Flush each buffer in batches
186
+ /**
187
+ * Send everything buffered, for every address.
188
+ *
189
+ * Drains run concurrently across addresses (they are independent buffers, and
190
+ * a shutdown should not pay for them one at a time) and every one is awaited
191
+ * before the first error is rethrown: an unreachable queue must not strand
192
+ * the others' messages.
193
+ */
194
+ async flushAllBuffers({ deadlineMillis = null } = {}) {
195
+ const queueAddresses = new Set([...this.#buffers.keys(), ...this.#drains.keys()])
196
+ logger.log('BufferManager.flushAllBuffers', { bufferCount: queueAddresses.size, pendingFlushes: this.#drains.size })
197
+
198
+ const drains = []
161
199
  for (const queueAddress of queueAddresses) {
162
- await this.flushBuffer(queueAddress)
200
+ const buffer = this.#buffers.get(queueAddress)
201
+ if (buffer) buffer.cancelTimer()
202
+ const drain = this.#startDrain(queueAddress, deadlineMillis)
203
+ if (drain) drains.push(drain)
163
204
  }
164
-
165
- logger.log('BufferManager.flushAllBuffers', { status: 'completed' })
205
+
206
+ const outcomes = await Promise.allSettled(drains)
207
+ const failure = outcomes.find(outcome => outcome.status === 'rejected')
208
+
209
+ logger.log('BufferManager.flushAllBuffers', { status: failure ? 'failed' : 'completed' })
210
+ if (failure) throw failure.reason
166
211
  }
167
212
 
168
- async #waitForPendingFlushes() {
169
- if (this.#pendingFlushes.size === 0) return
170
-
171
- logger.debug('BufferManager.waitForPendingFlushes', { count: this.#pendingFlushes.size })
172
- await Promise.all(Array.from(this.#pendingFlushes))
173
- logger.debug('BufferManager.waitForPendingFlushes', { status: 'completed' })
213
+ async #waitForDrains() {
214
+ if (this.#drains.size === 0) return
215
+ logger.debug('BufferManager.waitForDrains', { count: this.#drains.size })
216
+ await Promise.allSettled([...this.#drains.values()].map(entry => entry.promise))
217
+ logger.debug('BufferManager.waitForDrains', { status: 'completed' })
174
218
  }
175
219
 
176
220
  getStats() {
@@ -189,12 +233,25 @@ export class BufferManager {
189
233
  oldestBufferAge,
190
234
  flushesPerformed: this.#flushCount
191
235
  }
192
-
236
+
193
237
  logger.log('BufferManager.getStats', stats)
194
238
  return stats
195
239
  }
196
240
 
241
+ /**
242
+ * Stop every buffer and discard what is left.
243
+ *
244
+ * Stopping wakes parked adds (they reject: their message was never buffered)
245
+ * and ends any retry loop, so this also unhangs a drain that was waiting out
246
+ * a broker outage. Anything still buffered here is lost -- which is why
247
+ * Queen.close() flushes with a deadline first and logs what remains.
248
+ */
197
249
  cleanup() {
250
+ this.#stopped = true
251
+ const unflushed = this.getStats().totalBufferedMessages
252
+ if (unflushed > 0) {
253
+ logger.error('BufferManager.cleanup', { unflushedMessages: unflushed, status: 'discarded' })
254
+ }
198
255
  logger.log('BufferManager.cleanup', { bufferCount: this.#buffers.size })
199
256
  for (const buffer of this.#buffers.values()) {
200
257
  buffer.cleanup()
@@ -1,7 +1,43 @@
1
1
  /**
2
- * Message buffer for a single queue
2
+ * Message buffer for a single queue/partition.
3
+ *
4
+ * This is the client-side linger: single pushes accumulate here and leave as
5
+ * one request once `messageCount` messages are waiting, or `timeMillis` has
6
+ * passed since the first one arrived. Two properties beyond that batching are
7
+ * load-bearing, and neither existed before 2026-08-20:
8
+ *
9
+ * - `maxSize` is a BLOCKING bound, not a hint. `add()` returns a promise that
10
+ * does not resolve while the buffer is full, so a producer that outruns the
11
+ * flush pipeline is paced down to the drain rate instead of growing the
12
+ * heap. Measured on the Go client, whose buffer had exactly this shape:
13
+ * filling at 1.46M msg/s against a 1.0M msg/s flush pipeline accumulated
14
+ * 20.9M messages (11.7 GB of RSS) in 45 seconds and lost every one of them
15
+ * at process exit, with ZERO client-side errors reported anywhere. The
16
+ * bounded version sustained 881,148 msg/s with exact send/receive parity
17
+ * (39,655,787 = 39,655,787) and 71 MB of RSS.
18
+ *
19
+ * - A batch that fails to send is put BACK at the front of the buffer, in
20
+ * order, and retried after `retryDelayMillis`. It is never dropped. Before
21
+ * this, the flusher took the batch out before the POST and only logged the
22
+ * failure, so up to `messageCount` messages vanished per failed request.
23
+ *
24
+ * Together those two turn a broker outage into blocked producers with bounded
25
+ * memory, instead of silent loss.
26
+ *
27
+ * BLOCKING IDIOM: JavaScript is single-threaded, so "block" cannot mean
28
+ * "occupy the thread" -- that would starve the very flush that frees the
29
+ * capacity being waited for. It means an awaitable gate: parked adds hold a
30
+ * promise that the flusher resolves after each drained batch (and `stop()`
31
+ * rejects). No spin loop, no setInterval poll: the event loop is free to run
32
+ * the flush while producers wait.
33
+ *
34
+ * Buffered messages still live only in this process's memory. A crash, or a
35
+ * `process.exit()` that skips `close()`, loses them -- buffering belongs on
36
+ * telemetry-shaped traffic, not on anything that must not be lost.
3
37
  */
4
38
 
39
+ import { BUFFER_DEFAULTS } from '../utils/defaults.js'
40
+
5
41
  export class MessageBuffer {
6
42
  #queueAddress
7
43
  #messages = []
@@ -10,15 +46,93 @@ export class MessageBuffer {
10
46
  #timer = null
11
47
  #firstMessageTime = null
12
48
  #flushing = false
49
+ #stopped = false
50
+ // Adds parked on the maxSize bound. Each entry can be woken (capacity freed)
51
+ // or failed (stop, or the caller's AbortSignal).
52
+ #waiters = []
53
+ // Counts adds that have been woken but have not resumed yet. JS resumes an
54
+ // awaiting function on a later microtask, so the waiter list is already empty
55
+ // while those adds are still in flight; without this counter BufferManager
56
+ // could drop the buffer entry out of its map in that window and the resumed
57
+ // add would append to an orphan nobody ever flushes.
58
+ #parked = 0
59
+ #stopWaiters = []
13
60
 
14
61
  constructor(queueAddress, options, flushCallback) {
15
62
  this.#queueAddress = queueAddress
16
- this.#options = options
63
+ this.#options = MessageBuffer.normalizeOptions(options)
17
64
  this.#flushCallback = flushCallback
18
65
  }
19
66
 
20
- add(formattedMessage) {
21
- // Set first message time if this is the first message
67
+ /**
68
+ * Fill in defaults and enforce the bound's invariants.
69
+ *
70
+ * `maxSize: 0` (or absent) means the DEFAULT bound, never "unbounded":
71
+ * unbounded is the defect this knob exists to close, so opting out of
72
+ * backpressure is deliberately not expressible. The floor keeps the bound
73
+ * sane when a caller sets a `messageCount` larger than their `maxSize` --
74
+ * a buffer that must block before it can even assemble one batch would
75
+ * deadlock against its own flush threshold.
76
+ */
77
+ static normalizeOptions(options) {
78
+ // The caller's raw options, NOT BUFFER_DEFAULTS spread over them: the bound
79
+ // is derived from whatever messageCount this buffer ended up with, so
80
+ // `buffer({ messageCount: 10 })` gets a bound of 40, not the 400 that suits
81
+ // the default batch of 100. Unknown keys are carried through untouched.
82
+ const provided = options || {}
83
+
84
+ const messageCount = provided.messageCount > 0 ? provided.messageCount : BUFFER_DEFAULTS.messageCount
85
+ const timeMillis = provided.timeMillis > 0 ? provided.timeMillis : BUFFER_DEFAULTS.timeMillis
86
+
87
+ let maxSize = provided.maxSize > 0 ? provided.maxSize : 4 * messageCount
88
+ if (maxSize < messageCount) maxSize = messageCount
89
+
90
+ const retryDelayMillis = provided.retryDelayMillis > 0
91
+ ? provided.retryDelayMillis
92
+ : BUFFER_DEFAULTS.retryDelayMillis
93
+
94
+ return { ...provided, messageCount, timeMillis, maxSize, retryDelayMillis }
95
+ }
96
+
97
+ /**
98
+ * Append one message, waiting for room if the buffer is at its bound.
99
+ *
100
+ * Resolves once the message is in the buffer. Rejects if the buffer is
101
+ * stopped while parked, or if `signal` aborts -- an add that could not be
102
+ * buffered must never look like a successful push.
103
+ *
104
+ * @param {object} formattedMessage
105
+ * @param {{ signal?: AbortSignal }} [opts]
106
+ */
107
+ async add(formattedMessage, { signal } = {}) {
108
+ if (this.#stopped) {
109
+ throw new Error(`Queen buffer ${this.#queueAddress} is stopped: message not buffered`)
110
+ }
111
+ if (signal?.aborted) throw abortReason(signal)
112
+
113
+ // BACKPRESSURE. Re-checked in a loop, not once: a broadcast wakes every
114
+ // parked add, and the first ones to resume can fill the room that was
115
+ // freed, so the rest have to park again.
116
+ while (this.#messages.length >= this.#options.maxSize && !this.#stopped) {
117
+ // Being at the bound means producers outran the flusher. Make sure one is
118
+ // actually running before parking -- the time-based flush may be a full
119
+ // `timeMillis` away, and nothing else will start it.
120
+ this.#triggerFlush()
121
+ // The counter is held across the resumption itself, not just the wait:
122
+ // see #parked for why the window between "woken" and "resumed" matters.
123
+ this.#parked++
124
+ try {
125
+ await this.#waitForCapacity(signal)
126
+ } finally {
127
+ this.#parked--
128
+ }
129
+ if (signal?.aborted) throw abortReason(signal)
130
+ }
131
+
132
+ if (this.#stopped) {
133
+ throw new Error(`Queen buffer ${this.#queueAddress} stopped while waiting for capacity: message not buffered`)
134
+ }
135
+
22
136
  if (this.#messages.length === 0) {
23
137
  this.#firstMessageTime = Date.now()
24
138
  this.#startTimer()
@@ -26,77 +140,146 @@ export class MessageBuffer {
26
140
 
27
141
  this.#messages.push(formattedMessage)
28
142
 
29
- // Check if we should flush based on size
30
143
  if (this.#messages.length >= this.#options.messageCount) {
31
144
  this.#triggerFlush()
32
145
  }
33
146
  }
34
147
 
148
+ #waitForCapacity(signal) {
149
+ return new Promise((resolve, reject) => {
150
+ const waiter = { resolve, reject, signal, onAbort: null }
151
+
152
+ waiter.settle = (fn, arg) => {
153
+ const index = this.#waiters.indexOf(waiter)
154
+ if (index !== -1) this.#waiters.splice(index, 1)
155
+ if (waiter.onAbort) signal.removeEventListener('abort', waiter.onAbort)
156
+ fn(arg)
157
+ }
158
+
159
+ if (signal) {
160
+ waiter.onAbort = () => waiter.settle(reject, abortReason(signal))
161
+ signal.addEventListener('abort', waiter.onAbort, { once: true })
162
+ }
163
+
164
+ this.#waiters.push(waiter)
165
+ })
166
+ }
167
+
168
+ /**
169
+ * Wake every parked add. Called by the flusher after a batch is definitively
170
+ * gone (the POST succeeded), not when the batch is merely taken out of the
171
+ * buffer: a batch that fails goes straight back, and waking on extraction
172
+ * would let producers refill against room that never actually freed.
173
+ */
174
+ wakeWaiters() {
175
+ const waiters = this.#waiters.slice()
176
+ for (const waiter of waiters) waiter.settle(waiter.resolve)
177
+ }
178
+
35
179
  #startTimer() {
36
180
  if (this.#timer) return // Timer already running
37
181
 
182
+ // Deliberately NOT unref()'d: a short script that pushes and returns must
183
+ // stay alive long enough for the time-based flush to fire.
38
184
  this.#timer = setTimeout(() => {
185
+ this.#timer = null
39
186
  this.#triggerFlush()
40
187
  }, this.#options.timeMillis)
41
188
  }
42
189
 
43
190
  #triggerFlush() {
44
- if (this.#flushing || this.#messages.length === 0) return
45
-
46
- // Clear timer
47
- if (this.#timer) {
48
- clearTimeout(this.#timer)
49
- this.#timer = null
50
- }
51
-
52
- // Trigger flush via callback
191
+ if (this.#flushing || this.#stopped || this.#messages.length === 0) return
53
192
  this.#flushCallback(this.#queueAddress)
54
193
  }
55
194
 
56
- extractMessages(batchSize = null) {
57
- // If no batch size specified, extract all messages
58
- if (batchSize === null || batchSize >= this.#messages.length) {
59
- const messages = [...this.#messages]
60
- this.#messages = []
61
- this.#firstMessageTime = null
62
- this.#flushing = false
63
-
64
- if (this.#timer) {
65
- clearTimeout(this.#timer)
66
- this.#timer = null
67
- }
195
+ /**
196
+ * Claim the right to flush this buffer. Returns false when a flush is already
197
+ * running: one drain loop per buffer, so a burst of adds past the threshold
198
+ * cannot start a second sender that would interleave batches out of order.
199
+ */
200
+ beginFlush() {
201
+ if (this.#flushing || this.#stopped || this.#messages.length === 0) return false
202
+ this.#flushing = true
203
+ this.cancelTimer()
204
+ return true
205
+ }
68
206
 
69
- return messages
70
- }
207
+ /**
208
+ * Release the flush claim. Anything still buffered (a batch put back by a
209
+ * failed send, or messages added while the drain was stopping) gets a fresh
210
+ * timer, so it cannot sit there unnoticed until the next add.
211
+ */
212
+ endFlush() {
213
+ this.#flushing = false
214
+ if (this.#messages.length > 0 && !this.#stopped) this.#startTimer()
215
+ }
71
216
 
72
- // Extract a batch of messages
217
+ /**
218
+ * Take up to `batchSize` messages off the front.
219
+ *
220
+ * `splice` returns a fresh array, so the batch does not alias the buffer's
221
+ * storage and putting it back cannot corrupt what is left behind. (The Go
222
+ * reference has to copy explicitly there -- its slices do alias.)
223
+ */
224
+ takeBatch(batchSize) {
73
225
  const messages = this.#messages.splice(0, batchSize)
74
-
75
- // If buffer is now empty, reset state
76
- if (this.#messages.length === 0) {
77
- this.#firstMessageTime = null
78
- this.#flushing = false
79
-
80
- if (this.#timer) {
81
- clearTimeout(this.#timer)
82
- this.#timer = null
83
- }
84
- }
85
-
226
+ if (this.#messages.length === 0) this.#firstMessageTime = null
86
227
  return messages
87
228
  }
88
229
 
89
- setFlushing(value) {
90
- this.#flushing = value
230
+ /**
231
+ * Put a failed batch back at the FRONT, preserving order: these messages were
232
+ * queued before everything still in the buffer, and a retry must not reorder
233
+ * a partition's lane. Occupancy can overshoot maxSize by this one batch --
234
+ * documented on BUFFER_DEFAULTS.maxSize.
235
+ */
236
+ restoreBatch(messages) {
237
+ if (messages.length === 0) return
238
+ this.#messages.unshift(...messages)
239
+ if (this.#firstMessageTime === null) this.#firstMessageTime = Date.now()
91
240
  }
92
241
 
93
- forceFlush() {
94
- // Immediately trigger flush, ignoring timers
95
- if (this.#timer) {
96
- clearTimeout(this.#timer)
97
- this.#timer = null
242
+ /**
243
+ * Sleep, but return early if the buffer is stopped. Used for the retry delay
244
+ * between attempts at a failed batch: a shutdown must not wait out a delay
245
+ * that only exists to pace a broker that is not answering.
246
+ */
247
+ sleepUnlessStopped(millis) {
248
+ if (this.#stopped || millis <= 0) return Promise.resolve()
249
+ return new Promise(resolve => {
250
+ let timer = null
251
+ const wake = () => {
252
+ if (timer) clearTimeout(timer)
253
+ const index = this.#stopWaiters.indexOf(wake)
254
+ if (index !== -1) this.#stopWaiters.splice(index, 1)
255
+ resolve()
256
+ }
257
+ timer = setTimeout(wake, millis)
258
+ this.#stopWaiters.push(wake)
259
+ })
260
+ }
261
+
262
+ /**
263
+ * Stop accepting messages and wake everything parked, so shutdown cannot
264
+ * hang. Parked adds REJECT rather than resolve: their message was never
265
+ * buffered, and reporting success for a message that was dropped on the floor
266
+ * is the failure mode this whole change exists to remove.
267
+ */
268
+ stop() {
269
+ if (this.#stopped) return
270
+ this.#stopped = true
271
+ this.cancelTimer()
272
+
273
+ const waiters = this.#waiters.slice()
274
+ for (const waiter of waiters) {
275
+ waiter.settle(
276
+ waiter.reject,
277
+ new Error(`Queen buffer ${this.#queueAddress} stopped while waiting for capacity: message not buffered`)
278
+ )
98
279
  }
99
- this.#triggerFlush()
280
+
281
+ const sleepers = this.#stopWaiters.splice(0, this.#stopWaiters.length)
282
+ for (const wake of sleepers) wake()
100
283
  }
101
284
 
102
285
  cancelTimer() {
@@ -115,18 +298,39 @@ export class MessageBuffer {
115
298
  return this.#options
116
299
  }
117
300
 
301
+ get isFlushing() {
302
+ return this.#flushing
303
+ }
304
+
305
+ get isStopped() {
306
+ return this.#stopped
307
+ }
308
+
309
+ /** True while any add is parked on the bound, or woken but not yet resumed. */
310
+ get hasParkedAdds() {
311
+ return this.#waiters.length > 0 || this.#parked > 0
312
+ }
313
+
118
314
  get firstMessageAge() {
119
315
  return this.#firstMessageTime ? Date.now() - this.#firstMessageTime : 0
120
316
  }
121
317
 
122
318
  cleanup() {
123
- if (this.#timer) {
124
- clearTimeout(this.#timer)
125
- this.#timer = null
126
- }
319
+ this.stop()
127
320
  this.#messages = []
128
321
  this.#firstMessageTime = null
129
322
  this.#flushing = false
130
323
  }
131
324
  }
132
325
 
326
+ /**
327
+ * The reason an AbortSignal carries, or a plain AbortError when the runtime
328
+ * (or the caller's controller) did not set one.
329
+ */
330
+ function abortReason(signal) {
331
+ if (signal.reason instanceof Error) return signal.reason
332
+ const error = new Error('Queen buffered push aborted while waiting for buffer capacity')
333
+ error.name = 'AbortError'
334
+ if (signal.reason !== undefined) error.cause = signal.reason
335
+ return error
336
+ }
@@ -138,6 +138,21 @@ export class QueueBuilder {
138
138
  return this
139
139
  }
140
140
 
141
+ /**
142
+ * Batch pushes client-side instead of sending each one.
143
+ *
144
+ * @param {object} options
145
+ * @param {number} [options.messageCount=100] - flush once this many messages are waiting
146
+ * @param {number} [options.timeMillis=1000] - flush this long after the first message arrives
147
+ * @param {number} [options.maxSize] - backpressure bound: past this many buffered
148
+ * messages `push()` WAITS for the flusher instead of growing the heap.
149
+ * Defaults to 4 x messageCount, floored at messageCount. There is no
150
+ * unbounded setting: unbounded is what lost 20.9M messages in the
151
+ * 2026-08-20 measurement.
152
+ * @param {number} [options.retryDelayMillis=250] - delay before retrying a batch
153
+ * whose POST failed. Failed batches are re-queued at the front of the
154
+ * buffer and retried, never dropped.
155
+ */
141
156
  buffer(options) {
142
157
  this.#bufferOptions = options
143
158
  return this
@@ -622,20 +637,42 @@ class PushBuilder {
622
637
  async #execute() {
623
638
  logger.log('PushBuilder.execute', { queue: this.#queueName, partition: this.#partition, count: this.#formattedItems.length, buffered: !!this.#bufferOptions })
624
639
 
625
- // Client-side buffering
640
+ // Client-side buffering. Awaited, one message at a time: addMessage is
641
+ // where the maxSize backpressure bound is applied, so a buffered push
642
+ // resolves only once every item is actually IN the buffer. Firing these
643
+ // off without awaiting would report success for messages the buffer never
644
+ // accepted -- the exact failure this bound exists to remove.
626
645
  if (this.#bufferOptions) {
627
- for (const item of this.#formattedItems) {
628
- const queueAddress = `${this.#queueName}/${this.#partition}`
629
- this.#bufferManager.addMessage(queueAddress, item, this.#bufferOptions)
646
+ const queueAddress = `${this.#queueName}/${this.#partition}`
647
+ const accepted = []
648
+
649
+ try {
650
+ for (const item of this.#formattedItems) {
651
+ await this.#bufferManager.addMessage(queueAddress, item, this.#bufferOptions)
652
+ accepted.push(item)
653
+ }
654
+ } catch (error) {
655
+ // The buffer refused this message (the client is closing, or the wait
656
+ // for capacity was aborted). Report the items that did NOT make it, the
657
+ // same way the immediate push below reports rejected items -- counting
658
+ // them as buffered would be the false success this bound removes.
659
+ const unbuffered = this.#formattedItems.slice(accepted.length)
660
+ logger.error('PushBuilder.execute', { status: 'not-buffered', count: unbuffered.length, error: error.message })
661
+ if (this.#onErrorCallback) {
662
+ await this.#onErrorCallback(unbuffered, error)
663
+ return null
664
+ }
665
+ throw error
630
666
  }
631
- const result = { buffered: true, count: this.#formattedItems.length }
632
-
633
- logger.log('PushBuilder.execute', { status: 'buffered', count: this.#formattedItems.length })
634
-
667
+
668
+ const result = { buffered: true, count: accepted.length }
669
+
670
+ logger.log('PushBuilder.execute', { status: 'buffered', count: accepted.length })
671
+
635
672
  if (this.#onSuccessCallback) {
636
- await this.#onSuccessCallback(this.#formattedItems)
673
+ await this.#onSuccessCallback(accepted)
637
674
  }
638
-
675
+
639
676
  return result
640
677
  }
641
678
 
@@ -75,6 +75,20 @@ export const POP_DEFAULTS = {
75
75
 
76
76
  export const BUFFER_DEFAULTS = {
77
77
  messageCount: 100, // Flush after 100 messages
78
- timeMillis: 1000 // Or flush after 1 second
78
+ timeMillis: 1000, // Or flush after 1 second
79
+ // Backpressure bound: once this many messages are waiting, a buffered push
80
+ // WAITS for the flusher to drain below it instead of growing the heap. 0 (or
81
+ // absent) means 4 x messageCount -- "unbounded" is deliberately not
82
+ // expressible, because unbounded was the defect. Measured motivation
83
+ // (2026-08-20): without a bound, a producer filling at 1.46M msg/s against a
84
+ // 1.0M msg/s flush pipeline accumulated 20.9M messages (11.7 GB) in 45
85
+ // seconds and lost every one of them at process exit, with zero client-side
86
+ // errors. The bound is approximate: a batch that fails to send is put back,
87
+ // so occupancy can briefly overshoot by up to one messageCount.
88
+ maxSize: 400,
89
+ // How long the flusher waits before retrying a batch whose POST failed. The
90
+ // batch is re-queued at the front of the buffer and retried until it lands
91
+ // (or the buffer is stopped) -- never dropped. 0 means 250.
92
+ retryDelayMillis: 250
79
93
  }
80
94
 
package/package.json CHANGED
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "queen-mq",
3
- "version": "1.0.5",
3
+ "version": "1.0.6",
4
4
  "type": "module",
5
5
  "description": "Partitioned message queue on PostgreSQL — broker client + fluent streaming SDK (windows, joins, gates) in one package",
6
6
  "main": "client-v2/index.js",
7
7
  "scripts": {
8
- "test": "node --test test-v2/streams-unit/configHash.test.js test-v2/streams-unit/operators.test.js test-v2/streams-unit/cycle.test.js test-v2/streams-unit/eventTime.test.js test-v2/streams-unit/ack.test.js test-v2/http-unit/retry429.test.js test-v2/http-unit/hostHeader.test.js test-v2/kv-unit/kvWire.test.js test-v2/kv-unit/timerWire.test.js test-v2/kv-unit/txnWire.test.js && node test-v2/run.js human",
9
- "test:unit": "node --test test-v2/streams-unit/configHash.test.js test-v2/streams-unit/operators.test.js test-v2/streams-unit/cycle.test.js test-v2/streams-unit/eventTime.test.js test-v2/streams-unit/ack.test.js test-v2/http-unit/retry429.test.js test-v2/http-unit/hostHeader.test.js test-v2/kv-unit/kvWire.test.js test-v2/kv-unit/timerWire.test.js test-v2/kv-unit/txnWire.test.js",
8
+ "test": "node --test test-v2/streams-unit/configHash.test.js test-v2/streams-unit/operators.test.js test-v2/streams-unit/cycle.test.js test-v2/streams-unit/eventTime.test.js test-v2/streams-unit/ack.test.js test-v2/http-unit/retry429.test.js test-v2/http-unit/hostHeader.test.js test-v2/kv-unit/kvWire.test.js test-v2/kv-unit/timerWire.test.js test-v2/kv-unit/txnWire.test.js test-v2/buffer-unit/buffer.test.js && node test-v2/run.js human",
9
+ "test:unit": "node --test test-v2/streams-unit/configHash.test.js test-v2/streams-unit/operators.test.js test-v2/streams-unit/cycle.test.js test-v2/streams-unit/eventTime.test.js test-v2/streams-unit/ack.test.js test-v2/http-unit/retry429.test.js test-v2/http-unit/hostHeader.test.js test-v2/kv-unit/kvWire.test.js test-v2/kv-unit/timerWire.test.js test-v2/kv-unit/txnWire.test.js test-v2/buffer-unit/buffer.test.js",
10
10
  "test:unit:e2e": "node --test test-v2/streams-unit/e2e.test.js",
11
11
  "test:integration": "node test-v2/run.js human",
12
12
  "test:streams": "node test-v2/run.js stream",
@@ -0,0 +1,404 @@
1
+ /**
2
+ * Client-side push buffer: backpressure and loss-free retry.
3
+ *
4
+ * These pin the 2026-08-20 fix, ported from the Go reference
5
+ * (clients/client-go/message_buffer.go). Two defects were measured there and
6
+ * were present in this SDK too:
7
+ *
8
+ * 1. add() appended to an unbounded array and returned. A producer filling at
9
+ * 1.46M msg/s against a 1.0M msg/s flush pipeline accumulated 20.9M
10
+ * messages (11.7 GB of RSS) in 45 seconds and lost every one at process
11
+ * exit, with zero client-side errors reported.
12
+ * 2. The flusher took a batch out of the buffer before the POST and dropped
13
+ * it on error, losing up to messageCount messages per failed request.
14
+ *
15
+ * No broker and no network: the sink is a fake object with a post() the test
16
+ * drives, which is the whole point -- the buffer's contract is about ordering,
17
+ * occupancy and retries, none of which need a server to observe.
18
+ */
19
+
20
+ import { describe, it } from 'node:test'
21
+ import assert from 'node:assert/strict'
22
+
23
+ import { QueueBuilder } from '../../client-v2/builders/QueueBuilder.js'
24
+ import { BufferManager } from '../../client-v2/buffer/BufferManager.js'
25
+ import { MessageBuffer } from '../../client-v2/buffer/MessageBuffer.js'
26
+ import { BUFFER_DEFAULTS } from '../../client-v2/utils/defaults.js'
27
+
28
+ const ADDRESS = 'orders/Default'
29
+
30
+ const item = n => ({ queue: 'orders', partition: 'Default', payload: { n }, transactionId: `t-${n}` })
31
+ const payloadNumbers = items => items.map(i => i.payload.n)
32
+
33
+ /** A sink whose every POST hangs until the test releases or fails it. */
34
+ function controlledSink() {
35
+ const sent = []
36
+ const attempts = []
37
+ const inflight = []
38
+
39
+ return {
40
+ sent,
41
+ attempts,
42
+ get inflightCount() { return inflight.length },
43
+ post(_path, body) {
44
+ const items = body.items
45
+ attempts.push(items)
46
+ return new Promise((resolve, reject) => {
47
+ inflight.push({ items, resolve, reject })
48
+ })
49
+ },
50
+ /** Complete the oldest in-flight POST successfully. */
51
+ release() {
52
+ const call = inflight.shift()
53
+ assert.ok(call, 'no POST in flight to release')
54
+ sent.push(...call.items)
55
+ call.resolve([])
56
+ },
57
+ /** Fail the oldest in-flight POST. */
58
+ reject(message = 'connect ECONNREFUSED') {
59
+ const call = inflight.shift()
60
+ assert.ok(call, 'no POST in flight to reject')
61
+ call.reject(new Error(message))
62
+ }
63
+ }
64
+ }
65
+
66
+ /**
67
+ * A sink that answers immediately, failing whenever `shouldFail(attemptIndex)`
68
+ * says so. Used for the parity runs, where the interesting question is what
69
+ * comes out the far end after N failures, not when each one happened.
70
+ */
71
+ function flakySink(shouldFail) {
72
+ const sent = []
73
+ let attempt = 0
74
+ return {
75
+ sent,
76
+ get attempts() { return attempt },
77
+ async post(_path, body) {
78
+ const index = attempt++
79
+ if (shouldFail(index)) throw new Error(`sink refused attempt ${index}`)
80
+ sent.push(...body.items)
81
+ return []
82
+ }
83
+ }
84
+ }
85
+
86
+ const tick = (ms = 0) => new Promise(resolve => setTimeout(resolve, ms))
87
+
88
+ /** Poll until `predicate()` holds, or fail the test. Cheaper than a fixed sleep
89
+ * on an idle machine and less flaky than one on a loaded machine. */
90
+ async function until(predicate, what, timeoutMs = 2000) {
91
+ const deadline = Date.now() + timeoutMs
92
+ while (Date.now() < deadline) {
93
+ if (predicate()) return
94
+ await tick(2)
95
+ }
96
+ assert.fail(`timed out waiting for: ${what}`)
97
+ }
98
+
99
+ /** Track whether a promise has settled, without awaiting it. */
100
+ function watch(promise) {
101
+ const state = { settled: false, value: undefined, error: undefined }
102
+ promise.then(
103
+ value => { state.settled = true; state.value = value },
104
+ error => { state.settled = true; state.error = error }
105
+ )
106
+ return state
107
+ }
108
+
109
+ describe('buffer options', () => {
110
+ it('resolves an absent or zero maxSize to a BOUND, never to unbounded', () => {
111
+ // Unbounded is the defect this knob closes, so it is deliberately not
112
+ // expressible: 0 means the default bound, not infinity.
113
+ assert.equal(MessageBuffer.normalizeOptions({ messageCount: 10 }).maxSize, 40)
114
+ assert.equal(MessageBuffer.normalizeOptions({ messageCount: 10, maxSize: 0 }).maxSize, 40)
115
+ assert.equal(MessageBuffer.normalizeOptions({ messageCount: 10, maxSize: -1 }).maxSize, 40)
116
+ assert.equal(MessageBuffer.normalizeOptions({}).maxSize, BUFFER_DEFAULTS.maxSize)
117
+ })
118
+
119
+ it('floors maxSize at messageCount', () => {
120
+ // A bound below the flush threshold would block the producer before the
121
+ // buffer could assemble the batch that unblocks it.
122
+ assert.equal(MessageBuffer.normalizeOptions({ messageCount: 100, maxSize: 10 }).maxSize, 100)
123
+ })
124
+
125
+ it('defaults retryDelayMillis rather than retrying in a hot loop', () => {
126
+ assert.equal(MessageBuffer.normalizeOptions({ messageCount: 10 }).retryDelayMillis, 250)
127
+ assert.equal(MessageBuffer.normalizeOptions({ retryDelayMillis: 0 }).retryDelayMillis, 250)
128
+ assert.equal(MessageBuffer.normalizeOptions({ retryDelayMillis: 5 }).retryDelayMillis, 5)
129
+ })
130
+
131
+ it('ships a bounded default', () => {
132
+ assert.equal(BUFFER_DEFAULTS.maxSize, 4 * BUFFER_DEFAULTS.messageCount)
133
+ })
134
+ })
135
+
136
+ describe('backpressure', () => {
137
+ it('parks an add at the bound and resumes it when the flusher drains', async () => {
138
+ const sink = controlledSink()
139
+ const manager = new BufferManager(sink)
140
+ const options = { messageCount: 2, timeMillis: 60000, maxSize: 4, retryDelayMillis: 5 }
141
+
142
+ // The first two adds trip the count threshold and start the drain, which
143
+ // takes them and then hangs on the sink. The next four pile up behind it.
144
+ for (let n = 0; n < 6; n++) {
145
+ await manager.addMessage(ADDRESS, item(n), options)
146
+ }
147
+ await until(() => sink.inflightCount === 1, 'the first batch to reach the sink')
148
+ assert.equal(manager.getStats().totalBufferedMessages, 4, 'the buffer should be sitting exactly at its bound')
149
+
150
+ // The seventh add has nowhere to go: it must WAIT, not grow the buffer.
151
+ const parked = watch(manager.addMessage(ADDRESS, item(6), options))
152
+ await tick(20)
153
+ assert.equal(parked.settled, false, 'add returned while the buffer was full: that is the unbounded defect')
154
+ assert.equal(manager.getStats().totalBufferedMessages, 4, 'buffer grew past maxSize while an add was parked')
155
+
156
+ // Draining a batch frees room, which is what wakes the parked add.
157
+ sink.release()
158
+ await until(() => parked.settled, 'the parked add to resume once capacity freed')
159
+ assert.equal(parked.error, undefined, 'the resumed add must succeed, not error')
160
+ assert.deepEqual(payloadNumbers(sink.sent), [0, 1], 'the drained batch went out in order')
161
+
162
+ manager.cleanup()
163
+ })
164
+
165
+ it('wakes parked adds on cleanup, and tells them the message was NOT buffered', async () => {
166
+ const sink = controlledSink()
167
+ const manager = new BufferManager(sink)
168
+ const options = { messageCount: 2, timeMillis: 60000, maxSize: 2, retryDelayMillis: 5 }
169
+
170
+ for (let n = 0; n < 4; n++) {
171
+ await manager.addMessage(ADDRESS, item(n), options)
172
+ }
173
+ await until(() => manager.getStats().totalBufferedMessages === 2, 'the buffer to reach its bound')
174
+
175
+ const parked = watch(manager.addMessage(ADDRESS, item(99), options))
176
+ await tick(20)
177
+ assert.equal(parked.settled, false, 'precondition: the add is parked')
178
+
179
+ // Shutdown must not leave a producer waiting forever...
180
+ manager.cleanup()
181
+ await until(() => parked.settled, 'cleanup() to wake the parked add')
182
+
183
+ // ...and must not tell it the message went somewhere. It did not.
184
+ assert.ok(parked.error instanceof Error, 'a parked add woken by cleanup must reject, not resolve')
185
+ assert.match(parked.error.message, /not buffered/)
186
+ })
187
+
188
+ it('fails a parked add when its AbortSignal fires, instead of reporting success', async () => {
189
+ const sink = controlledSink()
190
+ const manager = new BufferManager(sink)
191
+ const options = { messageCount: 2, timeMillis: 60000, maxSize: 2, retryDelayMillis: 5 }
192
+
193
+ for (let n = 0; n < 4; n++) {
194
+ await manager.addMessage(ADDRESS, item(n), options)
195
+ }
196
+ await until(() => manager.getStats().totalBufferedMessages === 2, 'the buffer to reach its bound')
197
+
198
+ const controller = new AbortController()
199
+ const parked = watch(manager.addMessage(ADDRESS, item(99), options, { signal: controller.signal }))
200
+ await tick(20)
201
+ assert.equal(parked.settled, false, 'precondition: the add is parked')
202
+
203
+ controller.abort()
204
+ await until(() => parked.settled, 'the abort to release the parked add')
205
+ assert.ok(parked.error instanceof Error, 'an aborted add must reject')
206
+ assert.equal(parked.error.name, 'AbortError')
207
+ assert.equal(manager.getStats().totalBufferedMessages, 2, 'an aborted add must not leave its message behind')
208
+
209
+ manager.cleanup()
210
+ })
211
+ })
212
+
213
+ describe('failed flushes', () => {
214
+ it('puts a failed batch back at the FRONT, in order, and retries it after retryDelayMillis', async () => {
215
+ const sink = controlledSink()
216
+ const manager = new BufferManager(sink)
217
+ const options = { messageCount: 3, timeMillis: 60000, maxSize: 12, retryDelayMillis: 40 }
218
+
219
+ for (let n = 0; n < 3; n++) {
220
+ await manager.addMessage(ADDRESS, item(n), options)
221
+ }
222
+ await until(() => sink.inflightCount === 1, 'the first attempt')
223
+
224
+ const failedAt = Date.now()
225
+ sink.reject()
226
+
227
+ // The batch is back in the buffer -- not logged and forgotten.
228
+ await until(() => manager.getStats().totalBufferedMessages === 3, 'the failed batch to be re-queued')
229
+ assert.equal(manager.getStats().flushesPerformed, 0, 'a POST that never landed must not count as a flush')
230
+
231
+ await until(() => sink.inflightCount === 1, 'the retry')
232
+ assert.ok(Date.now() - failedAt >= 35, 'the retry must wait out retryDelayMillis, not spin')
233
+ assert.deepEqual(payloadNumbers(sink.attempts[1]), [0, 1, 2], 'the retry must resend the same batch, in the same order')
234
+
235
+ sink.release()
236
+ await until(() => manager.getStats().totalBufferedMessages === 0, 'the retry to drain the buffer')
237
+ assert.deepEqual(payloadNumbers(sink.sent), [0, 1, 2])
238
+ assert.equal(manager.getStats().flushesPerformed, 1)
239
+
240
+ manager.cleanup()
241
+ })
242
+
243
+ it('keeps a re-queued batch ahead of messages added while it was failing', async () => {
244
+ // Ordering within a partition is the product's headline promise: a retry
245
+ // that landed behind newer messages would reorder the lane.
246
+ const sink = controlledSink()
247
+ const manager = new BufferManager(sink)
248
+ const options = { messageCount: 2, timeMillis: 60000, maxSize: 20, retryDelayMillis: 10 }
249
+
250
+ for (let n = 0; n < 2; n++) {
251
+ await manager.addMessage(ADDRESS, item(n), options)
252
+ }
253
+ await until(() => sink.inflightCount === 1, 'the first attempt')
254
+ sink.reject()
255
+ await until(() => manager.getStats().totalBufferedMessages === 2, 're-queue')
256
+
257
+ for (let n = 2; n < 6; n++) {
258
+ await manager.addMessage(ADDRESS, item(n), options)
259
+ }
260
+
261
+ for (let batch = 0; batch < 3; batch++) {
262
+ await until(() => sink.inflightCount === 1, `attempt for batch ${batch}`)
263
+ sink.release()
264
+ }
265
+ await until(() => manager.getStats().totalBufferedMessages === 0, 'the buffer to drain')
266
+ assert.deepEqual(payloadNumbers(sink.sent), [0, 1, 2, 3, 4, 5])
267
+
268
+ manager.cleanup()
269
+ })
270
+
271
+ it('loses nothing across a run with intermittent failures', async () => {
272
+ // The whole point, stated as count parity: every message a caller was told
273
+ // was buffered comes out of the sink exactly once, in order, however many
274
+ // POSTs failed on the way.
275
+ const total = 500
276
+ const sink = flakySink(attempt => attempt % 3 === 1)
277
+ const manager = new BufferManager(sink)
278
+ const options = { messageCount: 7, timeMillis: 50, maxSize: 21, retryDelayMillis: 1 }
279
+
280
+ for (let n = 0; n < total; n++) {
281
+ await manager.addMessage(ADDRESS, item(n), options)
282
+ }
283
+ await manager.flushAllBuffers()
284
+
285
+ assert.equal(sink.sent.length, total, `sent ${sink.sent.length} of ${total}: messages were dropped`)
286
+ assert.deepEqual(payloadNumbers(sink.sent), Array.from({ length: total }, (_, n) => n), 'order was not preserved')
287
+ assert.equal(manager.getStats().totalBufferedMessages, 0)
288
+ assert.ok(sink.attempts > total / options.messageCount, 'precondition: the sink actually failed some attempts')
289
+
290
+ manager.cleanup()
291
+ })
292
+
293
+ it('bounds an explicit flush by its deadline and says how much is still buffered', async () => {
294
+ // Background flushes retry forever rather than drop; a shutdown cannot.
295
+ const sink = flakySink(() => true)
296
+ const manager = new BufferManager(sink)
297
+ const options = { messageCount: 2, timeMillis: 60000, maxSize: 8, retryDelayMillis: 5 }
298
+
299
+ for (let n = 0; n < 4; n++) {
300
+ await manager.addMessage(ADDRESS, item(n), options)
301
+ }
302
+
303
+ await assert.rejects(
304
+ () => manager.flushAllBuffers({ deadlineMillis: 30 }),
305
+ error => {
306
+ assert.equal(error.queenUnflushedCount, 4, 'the error must report what was left unsent')
307
+ assert.match(error.message, /still buffered/)
308
+ return true
309
+ }
310
+ )
311
+ assert.equal(manager.getStats().totalBufferedMessages, 4, 'a flush that gave up must leave the messages in the buffer, not drop them')
312
+
313
+ manager.cleanup()
314
+ })
315
+ })
316
+
317
+ describe('drain loop', () => {
318
+ it('runs one drain per buffer, no matter how many adds trip the threshold', async () => {
319
+ // Two senders on one partition would interleave batches and reorder the
320
+ // lane; the flushing flag is what prevents that.
321
+ const sink = controlledSink()
322
+ const manager = new BufferManager(sink)
323
+ const options = { messageCount: 2, timeMillis: 60000, maxSize: 100, retryDelayMillis: 5 }
324
+
325
+ for (let n = 0; n < 10; n++) {
326
+ await manager.addMessage(ADDRESS, item(n), options)
327
+ }
328
+ await until(() => sink.inflightCount === 1, 'the first batch')
329
+ await tick(20)
330
+ assert.equal(sink.inflightCount, 1, 'a second drain started: batches can now interleave')
331
+ assert.equal(sink.attempts.length, 1)
332
+
333
+ manager.cleanup()
334
+ })
335
+
336
+ it('keeps separate buffers per queue/partition address', async () => {
337
+ const sink = controlledSink()
338
+ const manager = new BufferManager(sink)
339
+ const options = { messageCount: 100, timeMillis: 60000, maxSize: 400, retryDelayMillis: 5 }
340
+
341
+ await manager.addMessage('orders/eu', item(1), options)
342
+ await manager.addMessage('orders/us', item(2), options)
343
+ await manager.addMessage('other/eu', item(3), options)
344
+
345
+ const stats = manager.getStats()
346
+ assert.equal(stats.activeBuffers, 3)
347
+ assert.equal(stats.totalBufferedMessages, 3)
348
+
349
+ manager.cleanup()
350
+ })
351
+
352
+ it('keeps the public push API shape now that the add path can wait', async () => {
353
+ // addMessage became awaitable so backpressure could exist at all. The
354
+ // builder is what most callers actually touch, so pin its shape: a buffered
355
+ // push still resolves to { buffered, count } and still lands in the buffer.
356
+ // The builder is driven over the fake sink rather than a Queen, so the
357
+ // test owns the teardown: a real client's pending flush timer would keep
358
+ // the runner alive for the full timeMillis.
359
+ const sink = controlledSink()
360
+ const manager = new BufferManager(sink)
361
+ const builder = new QueueBuilder(null, sink, manager, 'orders')
362
+
363
+ const result = await builder
364
+ .partition('eu')
365
+ .buffer({ messageCount: 1000, timeMillis: 60000 })
366
+ .push([{ data: { n: 1 } }, { data: { n: 2 } }])
367
+
368
+ assert.deepEqual(result, { buffered: true, count: 2 })
369
+ const stats = manager.getStats()
370
+ assert.equal(stats.totalBufferedMessages, 2)
371
+ assert.equal(stats.activeBuffers, 1, 'one buffer per queue/partition, not one per push')
372
+
373
+ manager.cleanup()
374
+ })
375
+
376
+ it('reports the items a stopped buffer refused instead of counting them as pushed', async () => {
377
+ const sink = controlledSink()
378
+ const manager = new BufferManager(sink)
379
+ const builder = new QueueBuilder(null, sink, manager, 'orders')
380
+ manager.cleanup() // client already closing
381
+
382
+ // push() returns a thenable builder, not a Promise, so await it inside.
383
+ await assert.rejects(
384
+ async () => { await builder.buffer({ messageCount: 1000, timeMillis: 60000 }).push([{ data: { n: 1 } }]) },
385
+ /not buffered/
386
+ )
387
+ assert.equal(manager.getStats().totalBufferedMessages, 0, 'a closed client must not accumulate messages nothing will flush')
388
+ })
389
+
390
+ it('flushes a buffer that never reaches its count, on the timer', async () => {
391
+ // The branch every other test here avoids with a 60s timer, and the one a
392
+ // low-volume producer actually lives on.
393
+ const sink = flakySink(() => false)
394
+ const manager = new BufferManager(sink)
395
+ const options = { messageCount: 1000, timeMillis: 30, maxSize: 4000, retryDelayMillis: 5 }
396
+
397
+ await manager.addMessage(ADDRESS, item(1), options)
398
+ await manager.addMessage(ADDRESS, item(2), options)
399
+ assert.equal(manager.getStats().totalBufferedMessages, 2, 'two messages are nowhere near the threshold')
400
+
401
+ await until(() => sink.sent.length === 2, 'the time-based flush to fire')
402
+ manager.cleanup()
403
+ })
404
+ })