async-rabbitmq 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1422 @@
1
+ require "async"
2
+ require "async/condition"
3
+ require "async/semaphore"
4
+ require_relative "errors"
5
+
6
+ module AsyncRabbitMQ
7
+ # Represents an AMQP channel multiplexed over a Session's connection.
8
+ #
9
+ # Duck-typed stream interface: #each (yields deliveries) and #write (publishes).
10
+ # Does NOT inherit Async::IO::Stream — a channel is logical, not physical IO.
11
+ class Channel
12
+ # A message published under confirms whose fate is unknown: the broker
13
+ # neither acked nor nacked it before the connection went. +payload+ and its
14
+ # routing are kept so it can be published again on another connection.
15
+ UnconfirmedMessage = Struct.new(:delivery_tag, :payload, :exchange, :routing_key, :options,
16
+ keyword_init: true)
17
+
18
+ # Where one publish call was addressed. Built once per call and shared by
19
+ # every message in a batch, so the send path allocates one small pair per
20
+ # message instead of a keyword struct and a copy of the options hash.
21
+ PublishContext = Struct.new(:exchange, :routing_key, :options)
22
+
23
+ # Fallback bound on the wait for channel.close-ok while resynchronising
24
+ # after an RPC timeout, used when the session has no rpc_timeout set.
25
+ RESYNC_TIMEOUT = 5
26
+
27
+ include Instrumented
28
+
29
+ attr_reader :channel_id, :pool_size
30
+
31
+ # How many times this channel has been opened on a connection: 0 until the
32
+ # first reopen, then one more for each connection recovery or #reopen.
33
+ #
34
+ # The broker restarts both delivery tags and publisher confirm tags at 1 on
35
+ # a reopened channel, so a tag only identifies a message together with the
36
+ # generation it was issued in. Deliveries carry theirs (see
37
+ # VersionedDeliveryTag); confirm tags returned by #basic_publish are plain
38
+ # integers, so code keeping its own confirm bookkeeping across a reconnect
39
+ # reads this.
40
+ attr_reader :delivery_generation
41
+
42
+ # +pool_size+ bounds the number of consumer-handler fibers that can run
43
+ # concurrently on this channel (Bunny-parity: default 1 = serialized).
44
+ # Resizable at runtime via #pool_size=; basic_qos adjusts it automatically
45
+ # when prefetch_count > 0 so the two stay coupled.
46
+ def initialize(channel_id, session, frame_io, frame_max:, logger:, pool_size: 1, rpc_timeout: nil)
47
+ @channel_id = channel_id
48
+ @session = session
49
+ @notifier = session.respond_to?(:notifier) ? session.notifier : nil
50
+ @frame_io = frame_io
51
+ @frame_max = frame_max
52
+ @logger = logger
53
+ @rpc_timeout = rpc_timeout # seconds a synchronous operation waits for its reply; nil = forever
54
+
55
+ @state = :closed
56
+ @queue = nil
57
+ @consumers = {} # consumer_tag => {queue_name:, block:, manual_ack:}
58
+ @each_waiters = {} # consumer_tag => Async::Condition, fibers blocked in #each
59
+ @return_handler = nil
60
+ @delivery_tag = 0
61
+ @pending_confirms = {} # delivery_tag => [frames, payload, PublishContext]
62
+ @nacked_tags = [] # tags the broker rejected since confirm_select (Bunny: nacked_set)
63
+ @only_acks = true # false once a nack arrives; read and reset by wait_for_confirms
64
+ @confirms_enabled = false
65
+ @tracking = false # confirm_select(tracking: true): backpressure + MessageNacked
66
+ @outstanding_limit = nil # max unconfirmed messages before basic_publish parks
67
+ @slot_condition = nil # publishers waiting for an outstanding slot
68
+ @nacked_this_cycle = [] # nacks since the last wait_for_confirms, for MessageNacked
69
+ @tx_mode = false
70
+ @prefetch = nil # last basic_qos settings, restored on reopen
71
+ @confirm_condition = nil
72
+ @flow_active = true
73
+ @on_cancel = nil
74
+ @on_error = nil
75
+ @on_handler_error = nil
76
+ @resyncing = false
77
+ @resync_condition = nil
78
+ # Incremented on every reopen (connection recovery or a single-channel
79
+ # reopen). Delivery tags are stamped with it so acks raised against a
80
+ # previous generation can be dropped instead of hitting a live message.
81
+ @delivery_generation = 0
82
+ @unsettled = {} # delivery tags delivered to a manual-ack consumer, not yet settled
83
+ @qos_warned = false
84
+ @mutex = Async::Semaphore.new(1)
85
+ @publish_sem = Async::Semaphore.new(1) # one publish's frames go out contiguously
86
+ @rpc_sem = Async::Semaphore.new(1) # one request/reply in flight per channel
87
+ @reply_condition = nil
88
+ @content_condition = nil
89
+ @pending_content = nil
90
+ @recovered_condition = nil # fibers parked while the connection recovers
91
+ @pool_size = validate_pool_size!(pool_size)
92
+ @pool_sem = Async::Semaphore.new(@pool_size)
93
+ end
94
+
95
+ # Resize the consumer-handler concurrency cap. Shrinking does not evict
96
+ # already-running handlers; new deliveries park until permits free up.
97
+ def pool_size=(n)
98
+ @pool_size = validate_pool_size!(n)
99
+ @pool_sem.limit = @pool_size
100
+ end
101
+
102
+ def open?
103
+ @state == :open
104
+ end
105
+
106
+ def closed?
107
+ @state == :closed
108
+ end
109
+
110
+ # True while the session is reconnecting; operations park until reopened.
111
+ def recovering?
112
+ @state == :recovering
113
+ end
114
+
115
+ # True while the channel is being reopened — after an RPC timeout, or
116
+ # through #reopen. Operations park until it is back, as they do during
117
+ # recovery: a publish issued now would either be dropped by the broker
118
+ # (the channel is closing) or overtake the unconfirmed messages being
119
+ # replayed onto the new one, and a request's reply would be discarded as
120
+ # the late one we are shedding.
121
+ def resyncing?
122
+ @state == :resyncing
123
+ end
124
+
125
+ def open
126
+ @queue = @frame_io.register_channel(@channel_id)
127
+ # Start dispatch task BEFORE waiting so it can process the OpenOk reply.
128
+ start_dispatch_task
129
+ rpc(AMQ::Protocol::Channel::Open.encode(@channel_id, ""), AMQ::Protocol::Channel::OpenOk, check_open: false)
130
+ @state = :open
131
+ instrument("channel.open") { { channel: @channel_id } }
132
+ self
133
+ end
134
+
135
+ def close
136
+ if recovering?
137
+ # Give the channel up rather than letting recovery reopen it.
138
+ drop!(NotOpenError.new("Channel #{@channel_id} closed during recovery"), reason: :user)
139
+ return
140
+ end
141
+ return unless open?
142
+ rpc(AMQ::Protocol::Channel::Close.encode(@channel_id, 200, "Goodbye", 0, 0), AMQ::Protocol::Channel::CloseOk)
143
+ @state = :closed
144
+ # Closing the channel cancelled its consumers on the broker side; an
145
+ # auto-delete queue that just lost its last one is gone with it.
146
+ forget_consumers_in_topology
147
+ instrument("channel.closed") { { channel: @channel_id, reason: :user } }
148
+ @session.channel_closed(@channel_id)
149
+ @queue&.push(nil) # wake dispatch_loop so it can detect :closed and exit
150
+ wake_each_waiters
151
+ end
152
+
153
+ # -------------------------------------------------------------------------
154
+ # Queue
155
+ # -------------------------------------------------------------------------
156
+
157
+ # Pass an empty string as +name+ to let the broker generate a unique name
158
+ # (returned in the Queue object). AMQP 0-9-1 spec §3.1.2.
159
+ def queue(name, passive: false, durable: false, exclusive: false, auto_delete: false, arguments: {})
160
+ resp = rpc(
161
+ AMQ::Protocol::Queue::Declare.encode(@channel_id, name, passive, durable, exclusive, auto_delete, false, arguments),
162
+ AMQ::Protocol::Queue::DeclareOk
163
+ )
164
+ q = Queue.new(resp.queue, resp.message_count, resp.consumer_count, self,
165
+ durable: durable, exclusive: exclusive, auto_delete: auto_delete)
166
+ unless passive
167
+ topology&.record_queue(@channel_id, resp.queue, durable: durable, exclusive: exclusive, auto_delete: auto_delete,
168
+ arguments: arguments, server_named: name.to_s.empty?, object: q)
169
+ end
170
+ q
171
+ end
172
+
173
+ # Server-named queue that lives for this connection only (exclusive, auto-delete).
174
+ def temporary_queue(**opts)
175
+ queue("", exclusive: true, auto_delete: true, **opts)
176
+ end
177
+
178
+ # Durable, non-exclusive, non-auto-delete queue of the given type
179
+ # (Queue::Types::CLASSIC, QUORUM or STREAM). A name is required: a
180
+ # server-named durable queue makes no sense. Durability, exclusivity and
181
+ # auto-delete are fixed; +passive+ and +arguments+ are honoured.
182
+ def durable_queue(name, type = Queue::Types::CLASSIC, passive: false, arguments: {})
183
+ if name.nil? || name.to_s.empty?
184
+ raise ArgumentError, "queue name must not be nil or empty (server-named durable queues make no sense)"
185
+ end
186
+ args = type.to_s == Queue::Types::CLASSIC ? arguments : { "x-queue-type" => type.to_s }.merge(arguments)
187
+ queue(name, passive: passive, durable: true, exclusive: false, auto_delete: false, arguments: args)
188
+ end
189
+
190
+ def quorum_queue(name, **opts)
191
+ durable_queue(name, Queue::Types::QUORUM, **opts)
192
+ end
193
+
194
+ # A RabbitMQ stream, usable over AMQP 0-9-1 as a durable queue. Consuming
195
+ # from it needs basic_qos and an "x-stream-offset" consumer argument.
196
+ def stream(name, **opts)
197
+ durable_queue(name, Queue::Types::STREAM, **opts)
198
+ end
199
+
200
+ def queue_delete(name, if_unused: false, if_empty: false)
201
+ rpc(AMQ::Protocol::Queue::Delete.encode(@channel_id, name, if_unused, if_empty, false), AMQ::Protocol::Queue::DeleteOk)
202
+ .tap { topology&.delete_queue(name) }
203
+ end
204
+
205
+ def queue_purge(name)
206
+ rpc(AMQ::Protocol::Queue::Purge.encode(@channel_id, name, false), AMQ::Protocol::Queue::PurgeOk)
207
+ end
208
+
209
+ def queue_bind(queue_name, exchange:, routing_key: "", arguments: {})
210
+ rpc(
211
+ AMQ::Protocol::Queue::Bind.encode(@channel_id, queue_name, exchange, routing_key, false, arguments),
212
+ AMQ::Protocol::Queue::BindOk
213
+ ).tap do
214
+ topology&.record_queue_binding(@channel_id, queue: queue_name, exchange: exchange,
215
+ routing_key: routing_key, arguments: arguments)
216
+ end
217
+ end
218
+
219
+ def queue_unbind(queue_name, exchange:, routing_key: "", arguments: {})
220
+ rpc(
221
+ AMQ::Protocol::Queue::Unbind.encode(@channel_id, queue_name, exchange, routing_key, arguments),
222
+ AMQ::Protocol::Queue::UnbindOk
223
+ ).tap do
224
+ topology&.delete_queue_binding(queue: queue_name, exchange: exchange, routing_key: routing_key, arguments: arguments)
225
+ end
226
+ end
227
+
228
+ # -------------------------------------------------------------------------
229
+ # Exchange
230
+ # -------------------------------------------------------------------------
231
+
232
+ def exchange(name, type: :direct, passive: false, durable: false, auto_delete: false, internal: false, arguments: {})
233
+ rpc(
234
+ AMQ::Protocol::Exchange::Declare.encode(@channel_id, name, type.to_s, passive, durable, auto_delete, internal, false, arguments),
235
+ AMQ::Protocol::Exchange::DeclareOk
236
+ )
237
+ unless passive
238
+ topology&.record_exchange(@channel_id, name, type, durable: durable, auto_delete: auto_delete,
239
+ internal: internal, arguments: arguments)
240
+ end
241
+ Exchange.new(name, type, self, durable: durable, auto_delete: auto_delete, internal: internal)
242
+ end
243
+
244
+ def direct(name, **opts)
245
+ exchange(name, type: :direct, **opts)
246
+ end
247
+
248
+ def fanout(name, **opts)
249
+ exchange(name, type: :fanout, **opts)
250
+ end
251
+
252
+ def topic(name, **opts)
253
+ exchange(name, type: :topic, **opts)
254
+ end
255
+
256
+ def headers(name, **opts)
257
+ exchange(name, type: :headers, **opts)
258
+ end
259
+
260
+ def default_exchange
261
+ Exchange.new("", :direct, self, durable: true, auto_delete: false, internal: false)
262
+ end
263
+
264
+ def exchange_delete(name, if_unused: false)
265
+ rpc(AMQ::Protocol::Exchange::Delete.encode(@channel_id, name, if_unused, false), AMQ::Protocol::Exchange::DeleteOk)
266
+ .tap { topology&.delete_exchange(name) }
267
+ end
268
+
269
+ def exchange_bind(destination:, source:, routing_key: "", arguments: {})
270
+ rpc(
271
+ AMQ::Protocol::Exchange::Bind.encode(@channel_id, destination, source, routing_key, false, arguments),
272
+ AMQ::Protocol::Exchange::BindOk
273
+ ).tap do
274
+ topology&.record_exchange_binding(@channel_id, source: source, destination: destination,
275
+ routing_key: routing_key, arguments: arguments)
276
+ end
277
+ end
278
+
279
+ def exchange_unbind(destination:, source:, routing_key: "", arguments: {})
280
+ rpc(
281
+ AMQ::Protocol::Exchange::Unbind.encode(@channel_id, destination, source, routing_key, false, arguments),
282
+ AMQ::Protocol::Exchange::UnbindOk
283
+ ).tap do
284
+ topology&.delete_exchange_binding(source: source, destination: destination, routing_key: routing_key, arguments: arguments)
285
+ end
286
+ end
287
+
288
+ # -------------------------------------------------------------------------
289
+ # Basic operations
290
+ # -------------------------------------------------------------------------
291
+
292
+ # Publish one message. Options: exchange:, routing_key:, mandatory:,
293
+ # persistent: (delivery_mode 2) and the standard AMQP properties
294
+ # (content_type:, content_encoding:, headers:, priority:, correlation_id:,
295
+ # reply_to:, expiration:, message_id:, timestamp:, type:, user_id:, app_id:)
296
+ # plus a raw properties: hash. Returns the confirm delivery tag when the
297
+ # channel is in confirm mode, nil otherwise.
298
+ def basic_publish(payload, exchange: "", routing_key: "", **opts)
299
+ bytes = encode_publish(payload, exchange: exchange, routing_key: routing_key, **opts)
300
+ context = @confirms_enabled ? publish_context(exchange, routing_key, opts) : nil
301
+ # One publish's frames must reach the wire contiguously and confirm tags
302
+ # must follow wire order, so the write and the tag assignment happen
303
+ # under one semaphore (write_frame can yield on a full queue or the
304
+ # connection.blocked gate).
305
+ tag = @publish_sem.acquire do
306
+ assert_open!
307
+ wait_for_outstanding_slot(1)
308
+ @frame_io.write_frame(bytes, publish: true)
309
+ reserve_confirm_tag(bytes, payload, context) if @confirms_enabled
310
+ end
311
+ instrument("message.published") do
312
+ { channel: @channel_id, exchange: exchange, routing_key: routing_key, count: 1,
313
+ bytes: bytes.bytesize, delivery_tag: tag }
314
+ end
315
+ tag
316
+ end
317
+
318
+ # Duck-typed #write for stream composability.
319
+ alias write basic_publish
320
+
321
+ # Publish many messages with one write. All payloads share +opts+ (see
322
+ # basic_publish). The frames are encoded into a single buffer and handed
323
+ # to the writer once; under confirms the tag range is reserved as a block
324
+ # and the tags are returned (nil otherwise). Batches of a few hundred to a
325
+ # few thousand messages give the best throughput (Bunny 3.0 parity).
326
+ def basic_publish_batch(payloads, exchange: "", routing_key: "", **opts)
327
+ raise ArgumentError, "payloads must be an Array of message bodies" unless payloads.is_a?(Array)
328
+ return nil if payloads.empty?
329
+
330
+ encoded = payloads.map { |p| encode_publish(p, exchange: exchange, routing_key: routing_key, **opts) }
331
+ tags = @publish_sem.acquire do
332
+ assert_open!
333
+ wait_for_outstanding_slot(encoded.size)
334
+ @frame_io.write_frame(encoded.join, publish: true)
335
+ if @confirms_enabled
336
+ context = publish_context(exchange, routing_key, opts)
337
+ payloads.each_with_index.map { |p, i| reserve_confirm_tag(encoded[i], p, context) }
338
+ end
339
+ end
340
+ instrument("message.published") do
341
+ { channel: @channel_id, exchange: exchange, routing_key: routing_key, count: encoded.size,
342
+ bytes: encoded.sum(&:bytesize), delivery_tag: tags&.last }
343
+ end
344
+ tags
345
+ end
346
+
347
+ # Synchronously fetch one message: [delivery_info, header, body], or nil if
348
+ # the queue is empty. Defaults to manual acknowledgement (as Bunny does):
349
+ # a message fetched and then dropped by the caller is requeued, not lost.
350
+ # Pass manual_ack: false to have the broker discard it on delivery.
351
+ def basic_get(queue_name, manual_ack: true)
352
+ msg, content = @rpc_sem.acquire do
353
+ assert_open!
354
+ @frame_io.write_frame(AMQ::Protocol::Basic::Get.encode(@channel_id, queue_name, !manual_ack).encode)
355
+ m = wait_for_any(AMQ::Protocol::Basic::GetOk, AMQ::Protocol::Basic::GetEmpty)
356
+ stamp_delivery_tag(m) if m.is_a?(AMQ::Protocol::Basic::GetOk)
357
+ # After GetOk the content header + body follow on this channel.
358
+ [m, m.is_a?(AMQ::Protocol::Basic::GetOk) ? wait_content : nil]
359
+ end
360
+ return nil if content.nil?
361
+
362
+ [msg, content[:header], content[:body]]
363
+ end
364
+
365
+ # Acknowledge a delivery. Returns true if the ack was sent, false if the
366
+ # tag came from an earlier connection (or an earlier life of this channel)
367
+ # and was dropped — see VersionedDeliveryTag.
368
+ def basic_ack(delivery_tag, multiple: false)
369
+ assert_open!
370
+ return false if stale_delivery_tag?(delivery_tag, "ack")
371
+ settle(delivery_tag, multiple: multiple)
372
+ @frame_io.write_frame(
373
+ AMQ::Protocol::Basic::Ack.encode(@channel_id, delivery_tag.to_i, multiple).encode
374
+ )
375
+ true
376
+ end
377
+
378
+ def basic_nack(delivery_tag, multiple: false, requeue: true)
379
+ assert_open!
380
+ return false if stale_delivery_tag?(delivery_tag, "nack")
381
+ settle(delivery_tag, multiple: multiple)
382
+ @frame_io.write_frame(
383
+ AMQ::Protocol::Basic::Nack.encode(@channel_id, delivery_tag.to_i, multiple, requeue).encode
384
+ )
385
+ true
386
+ end
387
+
388
+ def basic_reject(delivery_tag, requeue: true)
389
+ assert_open!
390
+ return false if stale_delivery_tag?(delivery_tag, "reject")
391
+ settle(delivery_tag)
392
+ @frame_io.write_frame(
393
+ AMQ::Protocol::Basic::Reject.encode(@channel_id, delivery_tag.to_i, requeue).encode
394
+ )
395
+ true
396
+ end
397
+
398
+ def basic_qos(prefetch_count:, prefetch_size: 0, global: false)
399
+ rpc(AMQ::Protocol::Basic::Qos.encode(@channel_id, prefetch_size, prefetch_count, global), AMQ::Protocol::Basic::QosOk)
400
+ @prefetch = { count: prefetch_count, size: prefetch_size, global: global } # restored on reopen
401
+ # Keep the handler pool coupled to prefetch: no point buffering 10 unacked
402
+ # messages at the broker if only 1 can run at a time. prefetch_count == 0
403
+ # means "unlimited" in AMQP; leave the pool alone so the user can still cap it.
404
+ self.pool_size = prefetch_count if prefetch_count > 0
405
+ end
406
+
407
+ # Start a consumer. The block runs in a new Async::Task per delivery,
408
+ # gated by the channel's pool_size semaphore so at most +pool_size+
409
+ # handlers run concurrently across all consumers on this channel.
410
+ # Returns the consumer tag.
411
+ def basic_consume(queue_name, consumer_tag: "", manual_ack: false, exclusive: false, arguments: {}, &block)
412
+ warn_unbounded_prefetch(queue_name)
413
+ resp = rpc(
414
+ AMQ::Protocol::Basic::Consume.encode(@channel_id, queue_name, consumer_tag, false, !manual_ack, exclusive, false, arguments),
415
+ AMQ::Protocol::Basic::ConsumeOk
416
+ )
417
+ @consumers[resp.consumer_tag] = { queue_name: queue_name, block: block, manual_ack: manual_ack }
418
+ topology&.record_consumer(resp.consumer_tag, queue_name)
419
+ instrument("consumer.registered") do
420
+ { channel: @channel_id, queue: queue_name, consumer_tag: resp.consumer_tag, manual_ack: manual_ack }
421
+ end
422
+ resp.consumer_tag
423
+ end
424
+
425
+ def basic_cancel(consumer_tag)
426
+ rpc(AMQ::Protocol::Basic::Cancel.encode(@channel_id, consumer_tag, false), AMQ::Protocol::Basic::CancelOk)
427
+ cancelled = @consumers.delete(consumer_tag)
428
+ topology&.delete_consumer(consumer_tag)
429
+ instrument("consumer.cancelled") do
430
+ { channel: @channel_id, consumer_tag: consumer_tag, queue: cancelled&.dig(:queue_name), reason: :client }
431
+ end
432
+ wake_each_waiters(consumer_tag)
433
+ end
434
+
435
+ # Ask the broker to redeliver all unacknowledged messages on this channel.
436
+ # RabbitMQ only supports requeue: true; requeue: false raises a channel error.
437
+ def basic_recover(requeue: true)
438
+ rpc(AMQ::Protocol::Basic::Recover.encode(@channel_id, requeue), AMQ::Protocol::Basic::RecoverOk)
439
+ end
440
+
441
+ # -------------------------------------------------------------------------
442
+ # Publisher confirms
443
+ # -------------------------------------------------------------------------
444
+
445
+ DEFAULT_OUTSTANDING_LIMIT = 1000
446
+
447
+ # Enable publisher confirms. With +tracking: true+ (Bunny 3.0 parity):
448
+ # basic_publish parks while +outstanding_limit+ messages are unconfirmed
449
+ # (default 1000, the sweet spot in Bunny's benchmarks), giving natural
450
+ # backpressure, and wait_for_confirms raises MessageNacked instead of
451
+ # returning false when the broker rejected a message.
452
+ def confirm_select(tracking: false, outstanding_limit: nil)
453
+ raise ArgumentError, "outstanding_limit requires tracking: true" if outstanding_limit && !tracking
454
+ if outstanding_limit && !(outstanding_limit.is_a?(Integer) && outstanding_limit.positive?)
455
+ raise ArgumentError, "outstanding_limit must be a positive Integer (got #{outstanding_limit.inspect})"
456
+ end
457
+
458
+ rpc(AMQ::Protocol::Confirm::Select.encode(@channel_id, false), AMQ::Protocol::Confirm::SelectOk)
459
+ @confirms_enabled = true
460
+ @tracking = tracking
461
+ @outstanding_limit = tracking ? (outstanding_limit || DEFAULT_OUTSTANDING_LIMIT) : nil
462
+ @delivery_tag = 0
463
+ @pending_confirms = {}
464
+ @nacked_tags = []
465
+ @nacked_this_cycle = []
466
+ @only_acks = true
467
+ @confirm_condition = Async::Condition.new
468
+ end
469
+
470
+ def tracking_confirms?
471
+ @tracking
472
+ end
473
+
474
+ attr_reader :outstanding_limit
475
+
476
+ # Block the current fiber until every published message has been acked or
477
+ # nacked by the broker. Returns true if all of them were acked since the
478
+ # previous call, false if at least one was nacked (see #nacked_tags).
479
+ # Raises ConnectionError if the session disconnects while waiting.
480
+ # +timeout+ in seconds bounds the whole wait; nil (the default) waits
481
+ # forever, as before. A broker that accepts a publish and then never
482
+ # confirms it otherwise parks the caller indefinitely.
483
+ def wait_for_confirms(timeout: nil)
484
+ deadline = timeout && (Process.clock_gettime(Process::CLOCK_MONOTONIC) + timeout)
485
+
486
+ until @pending_confirms.empty?
487
+ # Lazily (re)created: interrupt_wait! clears it on connection loss and a
488
+ # caller may arrive before reopen_after_recovery has re-selected confirms.
489
+ @confirm_condition ||= Async::Condition.new
490
+
491
+ if deadline
492
+ remaining = deadline - Process.clock_gettime(Process::CLOCK_MONOTONIC)
493
+ raise ConfirmTimeoutError.new(unconfirmed_tags: @pending_confirms.keys, channel_id: @channel_id) if remaining <= 0
494
+
495
+ begin
496
+ Async::Task.current.with_timeout(remaining) { @confirm_condition.wait }
497
+ rescue Async::TimeoutError
498
+ raise ConfirmTimeoutError.new(unconfirmed_tags: @pending_confirms.keys, channel_id: @channel_id)
499
+ end
500
+ else
501
+ @confirm_condition.wait
502
+ end
503
+
504
+ raise ConnectionError, "Session disconnected while waiting for confirms" unless open?
505
+ end
506
+ result = @only_acks
507
+ @only_acks = true
508
+ cycle_nacked = @nacked_this_cycle
509
+ @nacked_this_cycle = []
510
+ raise MessageNacked.new(nacked_tags: cycle_nacked, channel_id: @channel_id) if @tracking && !result
511
+ result
512
+ end
513
+
514
+ # Delivery tags the broker rejected with basic.nack since confirm_select.
515
+ def nacked_tags
516
+ @nacked_tags.dup
517
+ end
518
+
519
+ # Delivery tags published but not yet acked or nacked.
520
+ def unconfirmed_tags
521
+ @pending_confirms.keys
522
+ end
523
+
524
+ # The messages published under confirms that the broker has not yet acked
525
+ # or nacked, as UnconfirmedMessage records carrying the payload and where
526
+ # it was going. Unlike the encoded frames kept for replay on this channel,
527
+ # these can be republished anywhere — which is what a caller handed them by
528
+ # Cluster#on_node_down needs to do when the node they were on is gone.
529
+ def unconfirmed_messages
530
+ @pending_confirms.map do |tag, (_bytes, payload, context)|
531
+ UnconfirmedMessage.new(delivery_tag: tag, payload: payload,
532
+ exchange: context&.exchange, routing_key: context&.routing_key,
533
+ options: context ? context.options.dup : {})
534
+ end
535
+ end
536
+
537
+ # -------------------------------------------------------------------------
538
+ # Transactions (tx.*): publishes and acks on this channel are held by the
539
+ # broker until tx_commit, or discarded by tx_rollback. A channel cannot be
540
+ # both transactional and in confirm mode (the broker rejects the switch).
541
+ # Transactional mode is restored after connection recovery; work that was
542
+ # uncommitted when the connection dropped is lost, as with any client.
543
+ # -------------------------------------------------------------------------
544
+
545
+ def tx_select
546
+ rpc(AMQ::Protocol::Tx::Select.encode(@channel_id), AMQ::Protocol::Tx::SelectOk)
547
+ @tx_mode = true
548
+ end
549
+
550
+ def tx_commit
551
+ rpc(AMQ::Protocol::Tx::Commit.encode(@channel_id), AMQ::Protocol::Tx::CommitOk)
552
+ end
553
+
554
+ def tx_rollback
555
+ rpc(AMQ::Protocol::Tx::Rollback.encode(@channel_id), AMQ::Protocol::Tx::RollbackOk)
556
+ end
557
+
558
+ def using_tx?
559
+ @tx_mode
560
+ end
561
+
562
+ # -------------------------------------------------------------------------
563
+ # Return handler
564
+ # -------------------------------------------------------------------------
565
+
566
+ def on_return(&block)
567
+ @return_handler = block
568
+ end
569
+
570
+ attr_reader :return_handler
571
+
572
+ # Register a callback invoked when the broker cancels a consumer server-side
573
+ # (e.g. the queue is deleted or an HA failover occurs).
574
+ # The block receives the consumer_tag that was cancelled.
575
+ def on_cancel(&block)
576
+ @on_cancel = block
577
+ end
578
+
579
+ # Register a callback invoked when the broker closes this channel due to an
580
+ # error (e.g. 404 queue not found). The block receives the channel and the
581
+ # AMQ::Protocol::Channel::Close method frame.
582
+ def on_error(&block)
583
+ @on_error = block
584
+ end
585
+
586
+ # Register a callback invoked when a consumer block raises. The block
587
+ # receives the exception, the Basic::Deliver method frame and the queue
588
+ # name. The delivery has already been nacked (requeue: false) if the
589
+ # consumer was registered with manual_ack: true.
590
+ def on_handler_error(&block)
591
+ @on_handler_error = block
592
+ end
593
+
594
+ # -------------------------------------------------------------------------
595
+ # Flow control
596
+ # -------------------------------------------------------------------------
597
+
598
+ def flow(active)
599
+ rpc(AMQ::Protocol::Channel::Flow.encode(@channel_id, active), AMQ::Protocol::Channel::FlowOk)
600
+ @flow_active = active
601
+ end
602
+
603
+ # -------------------------------------------------------------------------
604
+ # Duck-typed stream: #each yields deliveries
605
+ # -------------------------------------------------------------------------
606
+
607
+ # Consume from +queue_name+, running +block+ for every delivery, and block
608
+ # the calling fiber until the consumer goes away: the channel is closed,
609
+ # the broker cancels the consumer (e.g. the queue is deleted) or closes the
610
+ # channel. A connection loss with automatic recovery is transparent: the
611
+ # consumer is re-registered and #each keeps waiting.
612
+ def each(queue_name, manual_ack: false, &block)
613
+ assert_open!
614
+ tag = basic_consume(queue_name, manual_ack: manual_ack, &block)
615
+ cond = @each_waiters[tag] = Async::Condition.new
616
+ cond.wait
617
+ nil
618
+ ensure
619
+ if tag
620
+ @each_waiters.delete(tag)
621
+ # Cancelled early (e.g. the calling task was stopped): tidy the consumer.
622
+ basic_cancel(tag) rescue nil if open? && @consumers.key?(tag)
623
+ end
624
+ end
625
+
626
+ # -------------------------------------------------------------------------
627
+ # Internal: connection state transitions driven by Session
628
+ # -------------------------------------------------------------------------
629
+
630
+ # The connection was lost and recovery is starting. Waits in flight fail
631
+ # with +error+; new publishes and RPCs park until #reopen_after_recovery
632
+ # (see #assert_open!); consumers are re-registered on reopen.
633
+ def mark_recovering!(error)
634
+ @state = :recovering
635
+ interrupt_wait!(error)
636
+ end
637
+
638
+ # Give the channel up while the session is recovering, instead of letting
639
+ # recovery reopen it: everything waiting or parked raises +error+, #each
640
+ # callers return, the consumers are forgotten and the id goes back.
641
+ def drop!(error, reason: :dropped)
642
+ mark_closed!(error)
643
+ forget_consumers_in_topology
644
+ @session.channel_closed(@channel_id)
645
+ instrument("channel.closed") { { channel: @channel_id, reason: reason } }
646
+ end
647
+
648
+ # The connection is gone for good: recovery disabled, exhausted, refused,
649
+ # or the session was closed. Everything waiting or parked raises +error+
650
+ # and #each callers return.
651
+ def mark_closed!(error)
652
+ @state = :closed
653
+ interrupt_wait!(error)
654
+ resume_parked!(error)
655
+ wake_each_waiters
656
+ end
657
+
658
+ # Unblock any fibers waiting on reply, content or confirm conditions
659
+ # (used on connection loss, and again if the connection dies mid-recovery).
660
+ def interrupt_wait!(error = nil)
661
+ error ||= ConnectionError.new(code: 0, text: "Connection lost during recovery")
662
+ @reply_condition&.signal(error)
663
+ @reply_condition = nil
664
+ @content_condition&.signal(error)
665
+ @content_condition = nil
666
+ @confirm_condition&.signal(error)
667
+ @confirm_condition = nil
668
+ release_outstanding_slots(error)
669
+ @queue&.push(nil) rescue nil
670
+ rescue => e
671
+ # ignore — best-effort unblock
672
+ end
673
+
674
+ # -------------------------------------------------------------------------
675
+ # Internal: called by Session after reconnect
676
+ # -------------------------------------------------------------------------
677
+
678
+ # Internal, after Session has reopened every channel and replayed the
679
+ # topology: release parked publishers/RPCs first (they may hold the
680
+ # semaphores basic_consume needs), then re-register the consumers.
681
+ def finish_recovery!
682
+ # After the session has replayed the topology, so a message addressed to a
683
+ # queue or exchange the broker lost is not sent into the void before it is
684
+ # re-declared. Before the channel goes :open, so that a new publisher
685
+ # cannot slip in if the replay yields on a full write queue and take a
686
+ # delivery tag that no longer matches its place on the wire.
687
+ republish_unconfirmed if @confirms_enabled
688
+ @state = :open
689
+ resume_parked!(:open)
690
+ re_register_consumers
691
+ end
692
+
693
+ # Internal, topology recovery. A re-declare the broker rejects closes the
694
+ # channel; the failure is logged and the channel reopened so the remaining
695
+ # entities can still be recovered (Bunny 3.2 behaviour).
696
+ def recover_exchange(x)
697
+ recover_entity("exchange #{x.name}") do
698
+ send_and_wait(AMQ::Protocol::Exchange::Declare.encode(@channel_id, x.name, x.type, false, x.durable,
699
+ x.auto_delete, x.internal, false, x.arguments),
700
+ AMQ::Protocol::Exchange::DeclareOk)
701
+ end
702
+ end
703
+
704
+ def recover_queue(q)
705
+ recover_entity("queue #{q.name}") do
706
+ ok = send_and_wait(AMQ::Protocol::Queue::Declare.encode(@channel_id, q.server_named ? "" : q.name, false,
707
+ q.durable, q.exclusive, q.auto_delete, false, q.arguments),
708
+ AMQ::Protocol::Queue::DeclareOk)
709
+ @session.queue_renamed(q.name, ok.queue) if ok.queue != q.name
710
+ end
711
+ end
712
+
713
+ def recover_queue_binding(b)
714
+ recover_entity("binding #{b.exchange} -> #{b.queue}") do
715
+ send_and_wait(AMQ::Protocol::Queue::Bind.encode(@channel_id, b.queue, b.exchange, b.routing_key, false, b.arguments),
716
+ AMQ::Protocol::Queue::BindOk)
717
+ end
718
+ end
719
+
720
+ def recover_exchange_binding(b)
721
+ recover_entity("exchange binding #{b.source} -> #{b.destination}") do
722
+ send_and_wait(AMQ::Protocol::Exchange::Bind.encode(@channel_id, b.destination, b.source, b.routing_key, false, b.arguments),
723
+ AMQ::Protocol::Exchange::BindOk)
724
+ end
725
+ end
726
+
727
+ # Internal: a server-named queue this channel consumes from came back under
728
+ # a new name after topology recovery.
729
+ def rename_consumer_queue(old_name, new_name)
730
+ @consumers.each_value { |c| c[:queue_name] = new_name if c[:queue_name] == old_name }
731
+ end
732
+
733
+ # The consumers on this channel no longer exist on the broker (the channel
734
+ # is closed, by us or by it). @consumers itself is kept so that
735
+ # reopen(recover_consumers: true) can register them again, which records
736
+ # them again.
737
+ def forget_consumers_in_topology
738
+ return unless (registry = topology)
739
+
740
+ @consumers.each_key { |tag| registry.delete_consumer(tag) }
741
+ end
742
+
743
+ # Reopen a channel the broker closed (e.g. delivery-ack timeout, unknown
744
+ # delivery tag) on the same connection, keeping its id. Prefetch, confirm
745
+ # mode and transactional mode are restored, and messages still unconfirmed
746
+ # at the time of the close are re-published. The old consumers are dropped
747
+ # unless +recover_consumers+ is true. (Bunny 3.0 parity.)
748
+ def reopen(recover_consumers: false)
749
+ unless closed?
750
+ raise NotOpenError, "Channel #{@channel_id} is #{@state}; only a closed channel can be reopened"
751
+ end
752
+ @consumers.clear unless recover_consumers
753
+ # Reopened :resyncing, not :open: the replay below must reach the wire
754
+ # before any new publish, or a tag no longer matches its position on it.
755
+ @session.reopen_channel(self, state: :resyncing)
756
+ # Only this channel was closed, so the topology is intact and anything
757
+ # left unconfirmed can go straight back out.
758
+ republish_unconfirmed if @confirms_enabled
759
+ @state = :open
760
+ resume_parked!(:open)
761
+ re_register_consumers if recover_consumers
762
+ self
763
+ end
764
+
765
+ # Internal: (re)open this channel on +frame_io+ and restore its settings.
766
+ # Uses direct sends: nothing else can be on the wire for this channel yet,
767
+ # and a fiber parked inside #rpc may be holding @rpc_sem. During connection
768
+ # recovery the session passes state: :recovering so callers stay parked
769
+ # until the topology has been replayed (see #finish_recovery!).
770
+ def reopen_on(frame_io, state: :open)
771
+ @frame_io = frame_io
772
+ # The broker restarts delivery tags at 1 on a reopened channel, so any
773
+ # tag handed out before this point is now stale.
774
+ @delivery_generation += 1
775
+ # Content parked for a basic_get that never collected it belongs to the
776
+ # old channel; leaving it would hand it to the next basic_get.
777
+ @pending_content = nil
778
+ # Whatever was outstanding belonged to the old channel; the broker
779
+ # requeued it and will deliver it again with fresh tags.
780
+ @unsettled.clear
781
+ @queue = @frame_io.register_channel(@channel_id)
782
+ start_dispatch_task
783
+ send_and_wait(AMQ::Protocol::Channel::Open.encode(@channel_id, ""), AMQ::Protocol::Channel::OpenOk)
784
+ @state = state
785
+
786
+ if (p = @prefetch)
787
+ send_and_wait(AMQ::Protocol::Basic::Qos.encode(@channel_id, p[:size], p[:count], p[:global]), AMQ::Protocol::Basic::QosOk)
788
+ end
789
+ # Confirm mode is restored here, but the unconfirmed messages are NOT sent
790
+ # yet: the topology they are addressed to may not be back (see
791
+ # #finish_recovery!, and #reopen for the single-channel case).
792
+ if @confirms_enabled
793
+ send_and_wait(AMQ::Protocol::Confirm::Select.encode(@channel_id, false), AMQ::Protocol::Confirm::SelectOk)
794
+ end
795
+ send_and_wait(AMQ::Protocol::Tx::Select.encode(@channel_id), AMQ::Protocol::Tx::SelectOk) if @tx_mode
796
+ end
797
+
798
+ private
799
+
800
+ # -------------------------------------------------------------------------
801
+ # Delivery tags
802
+ # -------------------------------------------------------------------------
803
+
804
+ # Replace the broker's raw tag with one that also carries this channel's
805
+ # generation. Basic::Deliver and Basic::GetOk expose delivery_tag through
806
+ # attr_reader, so setting the ivar is enough for handlers to pick it up.
807
+ def stamp_delivery_tag(method)
808
+ raw = method.instance_variable_get(:@delivery_tag)
809
+ return if raw.is_a?(VersionedDeliveryTag)
810
+
811
+ method.instance_variable_set(:@delivery_tag,
812
+ VersionedDeliveryTag.new(raw, @delivery_generation))
813
+ end
814
+
815
+ # A tag from an earlier generation refers to a message on a connection (or
816
+ # a life of this channel) that no longer exists. The broker restarts tag
817
+ # numbering at 1, so sending it now would either ack an unrelated message
818
+ # or draw a 406 that closes the channel and takes its consumers with it.
819
+ def stale_delivery_tag?(delivery_tag, action)
820
+ return false unless delivery_tag.is_a?(VersionedDeliveryTag)
821
+ return false unless delivery_tag.stale?(@delivery_generation)
822
+
823
+ @logger.warn(
824
+ "Channel #{@channel_id}: dropped #{action} for delivery tag #{delivery_tag.to_i} from " \
825
+ "generation #{delivery_tag.generation} (now #{@delivery_generation}) — the message was " \
826
+ "redelivered on the new connection"
827
+ )
828
+ instrument("delivery_tag.stale") do
829
+ { channel: @channel_id, action: action, delivery_tag: delivery_tag.to_i,
830
+ generation: delivery_tag.generation, current_generation: @delivery_generation }
831
+ end
832
+ true
833
+ end
834
+
835
+ # Without a prefetch limit the broker sends as fast as it can and the
836
+ # dispatch loop starts a task per delivery. pool_size caps how many run at
837
+ # once, not how many exist, so memory grows with the queue depth rather
838
+ # than with the concurrency. Warned once per channel; basic_qos fixes it.
839
+ def warn_unbounded_prefetch(queue_name)
840
+ return if @qos_warned
841
+ return if @prefetch && @prefetch[:count].to_i > 0
842
+
843
+ @qos_warned = true
844
+ @logger.warn(
845
+ "Channel #{@channel_id}: consuming from #{queue_name} without basic_qos. " \
846
+ "The broker will send the whole queue as fast as it can and one task is " \
847
+ "created per delivery, so memory tracks queue depth. Call " \
848
+ "basic_qos(prefetch_count: n) before basic_consume."
849
+ )
850
+ end
851
+
852
+ # Note a delivery as settled so nothing acks or nacks it a second time.
853
+ # +multiple+ settles every outstanding tag up to and including this one.
854
+ def settle(delivery_tag, multiple: false)
855
+ tag = delivery_tag.to_i
856
+ if multiple
857
+ @unsettled.delete_if { |t, _| t <= tag }
858
+ else
859
+ @unsettled.delete(tag)
860
+ end
861
+ end
862
+
863
+ # False once the handler (or anyone else) has acked, nacked or rejected it.
864
+ def unsettled?(delivery_tag)
865
+ @unsettled.key?(delivery_tag.to_i)
866
+ end
867
+
868
+ # A consumer block raised. Without this the delivery stays unacked for the
869
+ # life of the connection and, once prefetch messages are stuck that way,
870
+ # the consumer stops receiving anything at all.
871
+ def handle_consumer_error(error, method, entry)
872
+ @logger.error(
873
+ "Channel #{@channel_id}: consumer #{method.consumer_tag} raised " \
874
+ "#{error.class}: #{error.message}"
875
+ )
876
+ instrument("consumer.error") do
877
+ { channel: @channel_id, consumer_tag: method.consumer_tag, queue: entry[:queue_name],
878
+ error: error.class.name, message: error.message }
879
+ end
880
+
881
+ # Only a manual-ack delivery the handler has not already settled. A
882
+ # handler that acks and then raises in whatever follows would otherwise
883
+ # be nacking a tag the broker has already resolved: that is a 406, which
884
+ # closes the channel and takes its consumers with it.
885
+ if entry[:manual_ack] && unsettled?(method.delivery_tag)
886
+ begin
887
+ basic_nack(method.delivery_tag, requeue: false)
888
+ rescue => e
889
+ @logger.error("Channel #{@channel_id}: could not nack after consumer error: #{e.class}: #{e.message}")
890
+ end
891
+ end
892
+
893
+ return unless @on_handler_error
894
+
895
+ begin
896
+ @on_handler_error.call(error, method, entry[:queue_name])
897
+ rescue => e
898
+ @logger.error("Channel #{@channel_id}: on_handler_error hook raised #{e.class}: #{e.message}")
899
+ end
900
+ end
901
+
902
+ # -------------------------------------------------------------------------
903
+ # Frame dispatch
904
+ # -------------------------------------------------------------------------
905
+
906
+ def start_dispatch_task
907
+ @session.spawn_background { dispatch_loop }
908
+ end
909
+
910
+ def dispatch_loop
911
+ loop do
912
+ msg = @queue.pop
913
+ break if msg.nil?
914
+ handle_message(msg)
915
+ end
916
+ rescue => e
917
+ @logger.error("Channel #{@channel_id} dispatch error: #{e.class}: #{e.message}")
918
+ end
919
+
920
+ def handle_message(msg)
921
+ type, *rest = msg
922
+
923
+ case type
924
+ when :method
925
+ handle_method(rest[0])
926
+ when :content
927
+ # Content for a basic.get: hand it straight to the waiting fiber, or
928
+ # park it if the getter has not reached wait_content yet. It must never
929
+ # be left behind once consumed, or the next basic_get would receive
930
+ # the previous message's body.
931
+ content = { header: rest[0], body: rest[1] }
932
+ if (cond = @content_condition)
933
+ @content_condition = nil
934
+ cond.signal(content)
935
+ else
936
+ @pending_content = content
937
+ end
938
+ when :heartbeat
939
+ # connection-level, ignore at channel layer
940
+ end
941
+ end
942
+
943
+ def handle_method(method)
944
+ # While resynchronising, everything except the close-ok we are waiting
945
+ # for belongs to the request that timed out. Dropping it here is the
946
+ # point of the reopen: it must not reach the next caller.
947
+ if @resync_condition
948
+ case method
949
+ when AMQ::Protocol::Channel::CloseOk
950
+ cond = @resync_condition
951
+ @resync_condition = nil
952
+ cond.signal(:closed)
953
+ when AMQ::Protocol::Basic::Ack
954
+ # Confirms still in flight for messages published before the close.
955
+ # Dropping them would strand their entries in @pending_confirms: they
956
+ # would be republished as duplicates, and wait_for_confirms would
957
+ # never be woken for them.
958
+ handle_confirm_ack(method)
959
+ when AMQ::Protocol::Basic::Nack
960
+ handle_confirm_nack(method)
961
+ when AMQ::Protocol::Basic::Deliver, AMQ::Protocol::Basic::Return
962
+ # Content frames follow on this channel; take them off with the
963
+ # method so the stream does not skew.
964
+ @queue.pop
965
+ @logger.debug("Channel #{@channel_id}: discarded #{method.class} while resynchronising")
966
+ else
967
+ @logger.warn("Channel #{@channel_id}: discarded #{method.class} while resynchronising")
968
+ end
969
+ return
970
+ end
971
+
972
+ case method
973
+ when AMQ::Protocol::Basic::Deliver
974
+ # Pop content directly — must not suspend dispatch_loop via wait_content.
975
+ # After Deliver, the broker sends header+body frames immediately on this channel.
976
+ content_msg = @queue.pop
977
+ _, header, body = content_msg
978
+ entry = @consumers[method.consumer_tag]
979
+ if entry
980
+ stamp_delivery_tag(method)
981
+ @unsettled[method.delivery_tag.to_i] = true if entry[:manual_ack]
982
+ Async do
983
+ @pool_sem.acquire do
984
+ started = instrument_clock
985
+ begin
986
+ entry[:block].call(method, header, body)
987
+ rescue => e
988
+ handle_consumer_error(e, method, entry)
989
+ ensure
990
+ if started
991
+ instrument("message.consumed") do
992
+ { channel: @channel_id, queue: entry[:queue_name], consumer_tag: method.consumer_tag,
993
+ bytes: body.to_s.bytesize, redelivered: method.redelivered,
994
+ duration: instrument_elapsed(started) }
995
+ end
996
+ end
997
+ end
998
+ end
999
+ end
1000
+ else
1001
+ @logger.warn("Delivery on channel #{@channel_id} for unknown consumer #{method.consumer_tag}")
1002
+ end
1003
+
1004
+ when AMQ::Protocol::Basic::Return
1005
+ # Same pattern: pop content directly.
1006
+ content_msg = @queue.pop
1007
+ _, header, body = content_msg
1008
+ instrument("message.returned") do
1009
+ { channel: @channel_id, exchange: method.exchange, routing_key: method.routing_key,
1010
+ code: method.reply_code, text: method.reply_text, bytes: body.to_s.bytesize }
1011
+ end
1012
+ if @return_handler
1013
+ Async { @return_handler.call(method, header, body) }
1014
+ else
1015
+ @logger.warn("Unhandled basic.return on channel #{@channel_id} — register on_return to handle")
1016
+ end
1017
+
1018
+ when AMQ::Protocol::Basic::Ack
1019
+ handle_confirm_ack(method)
1020
+
1021
+ when AMQ::Protocol::Basic::Nack
1022
+ handle_confirm_nack(method)
1023
+
1024
+ when AMQ::Protocol::Channel::Close
1025
+ handle_channel_close(method)
1026
+
1027
+ when AMQ::Protocol::Basic::Cancel
1028
+ # Server-initiated consumer cancel (e.g. queue deleted, HA failover).
1029
+ # Remove from @consumers so deliveries are no longer dispatched to a dead block.
1030
+ entry = @consumers.delete(method.consumer_tag)
1031
+ topology&.delete_consumer(method.consumer_tag)
1032
+ if entry
1033
+ @logger.warn("Channel #{@channel_id}: broker cancelled consumer #{method.consumer_tag}")
1034
+ instrument("consumer.cancelled") do
1035
+ { channel: @channel_id, consumer_tag: method.consumer_tag, queue: entry[:queue_name], reason: :broker }
1036
+ end
1037
+ @on_cancel&.call(method.consumer_tag)
1038
+ end
1039
+ wake_each_waiters(method.consumer_tag)
1040
+ # No CancelOk to send for server-initiated cancel (no-wait is implicit).
1041
+
1042
+ when AMQ::Protocol::Channel::Flow
1043
+ # Server-initiated flow control — broker throttling this channel.
1044
+ @flow_active = method.active
1045
+ @frame_io.write_frame(AMQ::Protocol::Channel::FlowOk.encode(@channel_id, method.active).encode)
1046
+
1047
+ when AMQ::Protocol::Connection::Blocked,
1048
+ AMQ::Protocol::Connection::Unblocked
1049
+ # These arrive on channel 0 and are handled by the Session channel-0 monitor.
1050
+ # They should never reach a Channel object — log and ignore defensively.
1051
+ @logger.debug("Channel #{@channel_id}: ignoring connection-level #{method.class} frame")
1052
+
1053
+ else
1054
+ # Wake any fiber waiting on this method type
1055
+ @reply_condition&.signal(method)
1056
+ @reply_condition = nil
1057
+ # Yield so the newly-woken fiber (e.g. basic_consume registering its consumer)
1058
+ # can run before dispatch_loop processes the next queued message.
1059
+ Async::Task.current.yield
1060
+ end
1061
+ end
1062
+
1063
+ def handle_confirm_ack(method)
1064
+ @mutex.acquire do
1065
+ if method.multiple
1066
+ @pending_confirms.reject! { |tag, _| tag <= method.delivery_tag }
1067
+ else
1068
+ @pending_confirms.delete(method.delivery_tag)
1069
+ end
1070
+ @confirm_condition&.signal
1071
+ release_outstanding_slots
1072
+ end
1073
+ instrument("message.confirmed") do
1074
+ { channel: @channel_id, delivery_tag: method.delivery_tag, multiple: method.multiple, acked: true }
1075
+ end
1076
+ end
1077
+
1078
+ # A nack resolves the tag(s) like an ack does, but the rejection is recorded
1079
+ # so wait_for_confirms can report it instead of claiming success.
1080
+ def handle_confirm_nack(method)
1081
+ @mutex.acquire do
1082
+ rejected = if method.multiple
1083
+ @pending_confirms.keys.select { |tag| tag <= method.delivery_tag }
1084
+ else
1085
+ [method.delivery_tag]
1086
+ end
1087
+ rejected.each { |tag| @pending_confirms.delete(tag) }
1088
+ @nacked_tags.concat(rejected)
1089
+ @nacked_this_cycle.concat(rejected)
1090
+ @only_acks = false
1091
+ @confirm_condition&.signal
1092
+ release_outstanding_slots
1093
+ end
1094
+ instrument("message.confirmed") do
1095
+ { channel: @channel_id, delivery_tag: method.delivery_tag, multiple: method.multiple, acked: false }
1096
+ end
1097
+ end
1098
+
1099
+ def handle_channel_close(method)
1100
+ code = method.reply_code
1101
+ text = method.reply_text
1102
+ @frame_io.write_frame(
1103
+ AMQ::Protocol::Channel::CloseOk.encode(@channel_id).encode
1104
+ )
1105
+ @state = :closed
1106
+ forget_consumers_in_topology
1107
+ @session.channel_closed(@channel_id)
1108
+ instrument("channel.closed") { { channel: @channel_id, reason: :broker, code: code, text: text } }
1109
+ @on_error&.call(self, method)
1110
+
1111
+ error = if FrameIO::SOFT_ERROR_CODES.include?(code)
1112
+ ChannelError.new(code: code, text: text, channel_id: @channel_id, close_method: method)
1113
+ else
1114
+ ConnectionError.new(code: code, text: text)
1115
+ end
1116
+
1117
+ # Wake any waiting fiber with the error
1118
+ @reply_condition&.signal(error)
1119
+ @reply_condition = nil
1120
+ # Stop dispatch_loop
1121
+ @queue&.push(nil)
1122
+ wake_each_waiters
1123
+ end
1124
+
1125
+ # Release fibers blocked in #each for one consumer, or for all of them.
1126
+ def wake_each_waiters(tag = nil)
1127
+ waiters = tag ? [@each_waiters.delete(tag)].compact : @each_waiters.values.tap { @each_waiters.clear }
1128
+ waiters.each { |cond| cond.signal rescue nil }
1129
+ end
1130
+
1131
+ # -------------------------------------------------------------------------
1132
+ # Synchronous wait helpers
1133
+ # -------------------------------------------------------------------------
1134
+
1135
+ # Send one method frame and wait for its reply. Serialised per channel:
1136
+ # AMQP 0-9-1 replies carry no correlation id, so a second request in flight
1137
+ # on the same channel would be handed the first one's answer.
1138
+ def rpc(frame, *expected_classes, check_open: true)
1139
+ started = instrument_clock
1140
+ reply = rpc_without_instrumentation(frame, *expected_classes, check_open: check_open)
1141
+ if started
1142
+ instrument("channel.rpc") do
1143
+ # amq-protocol method classes report their AMQP name, e.g. "queue.declare-ok".
1144
+ { channel: @channel_id, method: reply.class.name, duration: instrument_elapsed(started) }
1145
+ end
1146
+ end
1147
+ reply
1148
+ end
1149
+
1150
+ def rpc_without_instrumentation(frame, *expected_classes, check_open: true)
1151
+ return send_and_wait(frame, *expected_classes) unless check_open
1152
+
1153
+ # Park before taking the semaphore so that reopen_after_recovery (which
1154
+ # bypasses it) is never blocked by a waiter holding it.
1155
+ wait_for_recovery! if recovering?
1156
+ timed_out = false
1157
+ begin
1158
+ @rpc_sem.acquire do
1159
+ assert_open!
1160
+ send_and_wait(frame, *expected_classes)
1161
+ end
1162
+ rescue RpcTimeoutError
1163
+ timed_out = true
1164
+ raise
1165
+ ensure
1166
+ # Outside the semaphore: the reply we gave up on may still be in
1167
+ # flight, and AMQP replies carry nothing to match them to a request,
1168
+ # so the next caller on this channel would collect it instead.
1169
+ resync_after_rpc_timeout if timed_out
1170
+ end
1171
+ end
1172
+
1173
+ # Close and reopen the channel after a reply never arrived, discarding
1174
+ # anything the broker sends for the abandoned request. Consumers are
1175
+ # re-registered, so a timeout costs a round trip rather than the channel's
1176
+ # deliveries. Runs outside @rpc_sem (re-registering consumers needs it).
1177
+ def resync_after_rpc_timeout
1178
+ return if @resyncing
1179
+ return unless open?
1180
+
1181
+ @resyncing = true
1182
+ # Park everyone else for the duration. The channel is about to be closed:
1183
+ # a publish issued now would go out after channel.close and be dropped by
1184
+ # the broker without a word, and another request's reply would be thrown
1185
+ # away as the late one we are shedding.
1186
+ @state = :resyncing
1187
+ @logger.warn(
1188
+ "Channel #{@channel_id}: reopening after an RPC timeout. A late reply carries nothing " \
1189
+ "to match it to its request, so it would be handed to the next caller on this channel."
1190
+ )
1191
+ instrument("channel.resync") { { channel: @channel_id, reason: :rpc_timeout } }
1192
+
1193
+ consumers = @consumers.dup
1194
+ cond = @resync_condition = Async::Condition.new
1195
+ @frame_io.write_frame(
1196
+ AMQ::Protocol::Channel::Close.encode(@channel_id, 200, "Resynchronising after RPC timeout", 0, 0).encode
1197
+ )
1198
+
1199
+ begin
1200
+ Async::Task.current.with_timeout(@rpc_timeout || RESYNC_TIMEOUT) { cond.wait }
1201
+ rescue Async::TimeoutError
1202
+ @logger.error("Channel #{@channel_id}: no channel.close-ok while resynchronising")
1203
+ ensure
1204
+ @resync_condition = nil
1205
+ end
1206
+
1207
+ @state = :closed
1208
+ @consumers.clear
1209
+ @session.reopen_channel(self, state: :resyncing)
1210
+ # reopen_on restores confirm mode but deliberately leaves the unconfirmed
1211
+ # messages alone: during connection recovery the topology they are
1212
+ # addressed to may not be back yet. Here only this channel went, so they
1213
+ # go straight out — and this is what resets @delivery_tag, which the
1214
+ # broker also restarts at 1 on the reopened channel. Without it the two
1215
+ # drift apart and every later confirm resolves the wrong publish.
1216
+ republish_unconfirmed if @confirms_enabled
1217
+ @state = :open
1218
+ @consumers.replace(consumers)
1219
+ re_register_consumers
1220
+ resume_parked!(:open)
1221
+ rescue => e
1222
+ @logger.error("Channel #{@channel_id}: could not resynchronise after an RPC timeout: #{e.class}: #{e.message}")
1223
+ mark_closed!(ChannelError.new("Channel could not be resynchronised after an RPC timeout: #{e.message}",
1224
+ channel_id: @channel_id))
1225
+ @session.channel_closed(@channel_id) rescue nil
1226
+ ensure
1227
+ @resyncing = false
1228
+ end
1229
+
1230
+ def send_and_wait(frame, *expected_classes)
1231
+ @frame_io.write_frame(frame.encode)
1232
+ wait_for_any(*expected_classes)
1233
+ end
1234
+
1235
+ def topology
1236
+ @session.respond_to?(:topology) ? @session.topology : nil
1237
+ end
1238
+
1239
+ def recover_entity(what)
1240
+ yield
1241
+ rescue ChannelError => e
1242
+ @logger.error("Channel #{@channel_id}: could not recover #{what}: #{e.message}")
1243
+ # The broker closed this channel; put it back so the rest can continue.
1244
+ @session.reopen_channel(self)
1245
+ end
1246
+
1247
+ # Encode one basic.publish (method + header + body frames) into a single
1248
+ # byte string, so a message always hits the write queue in one piece.
1249
+ def encode_publish(payload, exchange:, routing_key:, mandatory: false, persistent: false,
1250
+ content_type: nil, content_encoding: nil, headers: nil, priority: nil,
1251
+ correlation_id: nil, reply_to: nil, expiration: nil, message_id: nil,
1252
+ timestamp: nil, type: nil, user_id: nil, app_id: nil, properties: {})
1253
+ payload_bytes = payload.is_a?(String) ? payload.b : payload
1254
+ props = { delivery_mode: persistent ? 2 : 1 }
1255
+ props[:content_type] = content_type if content_type
1256
+ props[:content_encoding] = content_encoding if content_encoding
1257
+ props[:headers] = headers if headers
1258
+ props[:priority] = priority if priority
1259
+ props[:correlation_id] = correlation_id if correlation_id
1260
+ props[:reply_to] = reply_to if reply_to
1261
+ props[:expiration] = expiration if expiration
1262
+ props[:message_id] = message_id if message_id
1263
+ props[:timestamp] = timestamp if timestamp
1264
+ props[:type] = type if type
1265
+ props[:user_id] = user_id if user_id
1266
+ props[:app_id] = app_id if app_id
1267
+ props.merge!(properties)
1268
+
1269
+ # Basic::Publish.encode splits the body across frames of at most frame_max.
1270
+ AMQ::Protocol::Basic::Publish.encode(
1271
+ @channel_id, payload_bytes, props, exchange, routing_key, mandatory, false, @frame_max
1272
+ ).map(&:encode).join
1273
+ end
1274
+
1275
+ # Under confirms: take the next delivery tag and keep the encoded message
1276
+ # until the broker acks it, so it can be re-published after a reconnect.
1277
+ def reserve_confirm_tag(bytes, payload = nil, context = nil)
1278
+ @delivery_tag += 1
1279
+ @pending_confirms[@delivery_tag] = [bytes, payload, context]
1280
+ # The encoded frames carry this channel's id, so they can only be replayed
1281
+ # on this channel. Keep the payload and where it was going alongside, so a
1282
+ # caller handed these on a node failure can republish them somewhere else
1283
+ # (see #unconfirmed_messages). A pair, not a record: this runs for every
1284
+ # confirmed publish, and building the record here cost a quarter of the
1285
+ # send throughput. #unconfirmed_messages builds them on read instead,
1286
+ # which only happens when a node has actually gone.
1287
+ @delivery_tag
1288
+ end
1289
+
1290
+ # One per publish call. The caller keeps its headers hash and may go on
1291
+ # mutating it after the publish returns, so copy that too rather than only
1292
+ # the options around it.
1293
+ def publish_context(exchange, routing_key, options)
1294
+ copy = options ? options.dup : {}
1295
+ copy[:headers] = copy[:headers].dup if copy[:headers].is_a?(Hash)
1296
+ copy[:properties] = copy[:properties].dup if copy[:properties].is_a?(Hash)
1297
+ PublishContext.new(exchange, routing_key, copy).freeze
1298
+ end
1299
+
1300
+ # Confirm tracking backpressure: park until there is room for +needed+ more
1301
+ # unconfirmed messages under outstanding_limit (a batch larger than the
1302
+ # limit waits for the channel to be fully confirmed). Acks and nacks wake
1303
+ # the waiters; a connection loss or close fails them.
1304
+ def wait_for_outstanding_slot(needed)
1305
+ return unless @outstanding_limit
1306
+
1307
+ target = [@outstanding_limit - needed, 0].max
1308
+ while @pending_confirms.size > target
1309
+ @slot_condition ||= Async::Condition.new
1310
+ outcome = wait_with_timeout(@slot_condition) do
1311
+ @slot_condition = nil
1312
+ "No publisher confirm freed a slot within #{@rpc_timeout}s " \
1313
+ "(#{@pending_confirms.size} outstanding, limit #{@outstanding_limit}) on channel #{@channel_id}"
1314
+ end
1315
+ raise outcome if outcome.is_a?(Exception)
1316
+ assert_open!
1317
+ end
1318
+ end
1319
+
1320
+ def release_outstanding_slots(outcome = nil)
1321
+ cond = @slot_condition
1322
+ @slot_condition = nil
1323
+ cond&.signal(outcome)
1324
+ end
1325
+
1326
+ # Messages published under confirms whose ack never arrived are sent again
1327
+ # on the reopened channel with fresh delivery tags (numbering restarts at 1).
1328
+ # A message the broker had in fact accepted before the drop is delivered
1329
+ # twice: the usual at-least-once trade-off of confirms across a reconnect.
1330
+ def republish_unconfirmed
1331
+ pending = @pending_confirms.values
1332
+ @pending_confirms = {}
1333
+ @delivery_tag = 0
1334
+ @confirm_condition ||= Async::Condition.new # keep one an early waiter created
1335
+ pending.each do |entry|
1336
+ @frame_io.write_frame(entry[0], publish: true)
1337
+ @delivery_tag += 1
1338
+ @pending_confirms[@delivery_tag] = entry
1339
+ end
1340
+ @logger.info("Channel #{@channel_id}: re-published #{pending.size} unconfirmed message(s) after recovery") unless pending.empty?
1341
+ end
1342
+
1343
+ def wait_for_recovery!
1344
+ @recovered_condition ||= Async::Condition.new
1345
+ outcome = @recovered_condition.wait
1346
+ raise outcome if outcome.is_a?(Exception)
1347
+ end
1348
+
1349
+ def resume_parked!(outcome)
1350
+ cond = @recovered_condition
1351
+ @recovered_condition = nil
1352
+ cond&.signal(outcome)
1353
+ end
1354
+
1355
+ def wait_for_any(*expected_classes)
1356
+ condition = @reply_condition = Async::Condition.new
1357
+ result = wait_with_timeout(condition) do
1358
+ @reply_condition = nil if @reply_condition.equal?(condition)
1359
+ # amq-protocol method classes report their AMQP name, e.g. "queue.declare-ok".
1360
+ "No reply to #{expected_classes.map(&:name).join('/')} within #{@rpc_timeout}s on channel #{@channel_id}"
1361
+ end
1362
+
1363
+ raise result if result.is_a?(Exception)
1364
+
1365
+ unless expected_classes.any? { |c| result.is_a?(c) }
1366
+ raise ChannelError.new("Expected #{expected_classes.join(' or ')} but got #{result.class}",
1367
+ channel_id: @channel_id)
1368
+ end
1369
+ result
1370
+ end
1371
+
1372
+ def wait_content
1373
+ if @pending_content
1374
+ content = @pending_content
1375
+ @pending_content = nil
1376
+ return content
1377
+ end
1378
+
1379
+ condition = @content_condition = Async::Condition.new
1380
+ wait_with_timeout(condition) do
1381
+ @content_condition = nil if @content_condition.equal?(condition)
1382
+ "No message content received within #{@rpc_timeout}s on channel #{@channel_id}"
1383
+ end
1384
+ end
1385
+
1386
+ # Wait on +condition+, bounded by the channel's rpc_timeout. On expiry the
1387
+ # block tidies the waiter slot and returns the message for RpcTimeoutError.
1388
+ def wait_with_timeout(condition)
1389
+ return condition.wait unless @rpc_timeout
1390
+
1391
+ Async::Task.current.with_timeout(@rpc_timeout) { condition.wait }
1392
+ rescue Async::TimeoutError
1393
+ raise RpcTimeoutError, yield
1394
+ end
1395
+
1396
+ def re_register_consumers
1397
+ @consumers.each do |tag, entry|
1398
+ begin
1399
+ basic_consume(entry[:queue_name], consumer_tag: tag, manual_ack: entry[:manual_ack], &entry[:block])
1400
+ rescue => e
1401
+ # Typically 404: the queue is gone and was not recovered. The broker
1402
+ # closes the channel on that, so say so instead of failing silently.
1403
+ @logger.error("Channel #{@channel_id}: could not re-register consumer #{tag} on #{entry[:queue_name]}: #{e.class}: #{e.message}")
1404
+ end
1405
+ end
1406
+ end
1407
+
1408
+ # Raise unless the channel is open. While the connection is being recovered
1409
+ # the caller is parked instead and resumes once the channel is reopened, so
1410
+ # publishes and RPCs issued during an outage neither fail nor vanish; if
1411
+ # recovery is abandoned they raise the final error.
1412
+ def assert_open!
1413
+ wait_for_recovery! while recovering? || resyncing?
1414
+ raise NotOpenError, "Channel #{@channel_id} is not open" unless open?
1415
+ end
1416
+
1417
+ def validate_pool_size!(n)
1418
+ raise ArgumentError, "pool_size must be a positive Integer (got #{n.inspect})" unless n.is_a?(Integer) && n > 0
1419
+ n
1420
+ end
1421
+ end
1422
+ end