async-rabbitmq 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1154 @@
1
+ require "async"
2
+ require "async/condition"
3
+ require "async/semaphore"
4
+ require "socket"
5
+ require "openssl"
6
+ require "uri"
7
+ require "amq/uri"
8
+ require_relative "errors"
9
+ require_relative "frame_io"
10
+ require_relative "channel"
11
+ require_relative "sasl"
12
+
13
+ module AsyncRabbitMQ
14
+ # Represents one AMQP connection to a RabbitMQ broker.
15
+ #
16
+ # Recovery: exponential backoff with configurable interval, max interval,
17
+ # and retry limit. Re-registers consumers after reconnect.
18
+ #
19
+ # Pool interface: implements reusable?, viable?, concurrency, close for Async::Pool.
20
+ class Session
21
+ include Instrumented
22
+
23
+ CONNECT_TIMEOUT = 30 # seconds for AMQP handshake
24
+ RPC_TIMEOUT = 15 # seconds to wait for a synchronous channel reply (nil = forever)
25
+ RECOVERY_INITIAL = 1.0 # seconds
26
+ RECOVERY_MAX = 30.0 # seconds
27
+ RECOVERY_JITTER = 0.25 # ±25%
28
+ PROTOCOL_HEADER = "AMQP\x00\x00\x09\x01".b.freeze
29
+
30
+ attr_reader :host, :port, :vhost, :username, :addresses
31
+
32
+ # The TopologyRegistry of exchanges, queues and bindings declared through
33
+ # this session, replayed after a reconnect (see recover_topology:).
34
+ attr_reader :topology
35
+
36
+ # Structured events for metrics and tracing; see Notifier::EVENTS.
37
+ attr_reader :notifier
38
+
39
+ # Build a Session from one or more AMQP URI strings.
40
+ #
41
+ # Session.from_uri("amqp://user:pass@rabbit:5672/myvhost")
42
+ # Session.from_uri("amqps://rabbit/myvhost", tls_context: ctx)
43
+ # Session.from_uri("amqp://rabbit1:5672/vh", "amqp://rabbit2:5672/vh")
44
+ #
45
+ # When multiple URIs are given, credentials, vhost, and TLS settings are
46
+ # taken from the first URI. Each URI contributes a host:port pair to the
47
+ # +addresses+ list used for failover.
48
+ #
49
+ # Keyword arguments override anything parsed from the URI(s).
50
+ #
51
+ # The standard RabbitMQ URI query parameters are honoured: +heartbeat+,
52
+ # +connection_timeout+, +channel_max+, +auth_mechanism+ and, for amqps://,
53
+ # +verify+, +cacertfile+, +certfile+ and +keyfile+ (which build a
54
+ # +tls_context+ unless one is passed explicitly).
55
+ def self.from_uri(*uri_strings, **kwargs)
56
+ new(**options_from_uris(*uri_strings, **kwargs))
57
+ end
58
+
59
+ # The Session.new keyword arguments for one or more URIs; see from_uri.
60
+ # Cluster.from_uri builds its nodes from the same list.
61
+ def self.options_from_uris(*uri_strings, **kwargs)
62
+ raise ArgumentError, "at least one URI string is required" if uri_strings.empty?
63
+
64
+ first_opts = uri_options(uri_strings.first)
65
+ address_list = uri_strings.map do |u|
66
+ parsed = uri_options(u)
67
+ "#{parsed[:host] || 'localhost'}:#{parsed[:port]}"
68
+ end
69
+
70
+ # Connection-level settings come from the first URI; addresses from all.
71
+ opts = first_opts.except(:host, :port)
72
+ opts[:addresses] = address_list
73
+ opts.merge(kwargs)
74
+ end
75
+
76
+ # Translate an amqp:// or amqps:// URI into Session.new keyword arguments.
77
+ #
78
+ # Parsing is delegated to AMQ::URI from amq-protocol, which percent-decodes
79
+ # the credentials and vhost (so "pa+ss" stays "pa+ss") and validates the
80
+ # scheme, the single-segment vhost path and the TLS-only query parameters.
81
+ def self.uri_options(uri_string)
82
+ parsed = AMQ::URI.parse(uri_string)
83
+ query = ::URI.parse(uri_string).query
84
+ params = query ? ::URI.decode_www_form(query).to_h : {}
85
+
86
+ opts = { tls: !!parsed[:ssl], port: parsed[:port] }
87
+ opts[:host] = parsed[:host] if parsed[:host]
88
+ opts[:username] = parsed[:user] if parsed[:user]
89
+ opts[:password] = parsed[:pass] if parsed[:pass]
90
+ # "amqp://host/" yields an empty vhost; treat it as the default "/" rather
91
+ # than the (rarely intended) vhost literally named "".
92
+ opts[:vhost] = parsed[:vhost].empty? ? "/" : parsed[:vhost] if parsed.key?(:vhost)
93
+
94
+ opts[:heartbeat] = Integer(params["heartbeat"]) if params.key?("heartbeat")
95
+ opts[:connect_timeout] = Integer(params["connection_timeout"]) if params.key?("connection_timeout")
96
+ opts[:channel_max] = Integer(params["channel_max"]) if params.key?("channel_max")
97
+ opts[:auth_mechanism] = params["auth_mechanism"] if params.key?("auth_mechanism")
98
+
99
+ if opts[:tls] && %w[verify cacertfile certfile keyfile].any? { |k| params.key?(k) }
100
+ opts[:tls_context] = tls_context_from_uri(parsed, params)
101
+ end
102
+
103
+ opts
104
+ end
105
+
106
+ # Build an OpenSSL context from the TLS query parameters of an amqps:// URI,
107
+ # the same way the tls_cert:/tls_key:/tls_ca_certificates: options do.
108
+ # Peer verification stays on unless the URI says verify=false explicitly.
109
+ def self.tls_context_from_uri(parsed, params)
110
+ verify = params.key?("verify") ? params["verify"] != "false" : true
111
+ TLS.context(cert: parsed[:certfile], key: parsed[:keyfile], ca_certificates: parsed[:cacertfile],
112
+ verify_peer: verify)
113
+ end
114
+
115
+ private_class_method :uri_options, :tls_context_from_uri
116
+
117
+ # The canonical [host, port] list from the various input forms.
118
+ #
119
+ # addresses: ["rabbit1:5672", "rabbit2:5673"] -> [["rabbit1",5672], ["rabbit2",5673]]
120
+ # hosts: ["rabbit1", "rabbit2"], port: 5672 -> [["rabbit1",5672], ["rabbit2",5672]]
121
+ # host: "rabbit1", port: 5672 (default) -> [["rabbit1",5672]]
122
+ def self.address_list(host:, port:, hosts: nil, addresses: nil)
123
+ if addresses && !addresses.empty?
124
+ addresses.map do |addr|
125
+ h, p = addr.to_s.split(":", 2)
126
+ [h, p ? p.to_i : port]
127
+ end
128
+ elsif hosts && !hosts.empty?
129
+ hosts.map { |h| [h.to_s, port] }
130
+ else
131
+ [[host.to_s, port]]
132
+ end
133
+ end
134
+
135
+ def initialize(
136
+ host: "localhost",
137
+ port: 5672,
138
+ hosts: nil,
139
+ addresses: nil,
140
+ hosts_shuffle_strategy: :shuffle,
141
+ vhost: "/",
142
+ username: "guest",
143
+ password: "guest",
144
+ tls: false,
145
+ tls_context: nil,
146
+ tls_cert: nil,
147
+ tls_key: nil,
148
+ tls_ca_certificates: nil,
149
+ verify_peer: true,
150
+ tls_min_version: :TLS1_2,
151
+ heartbeat: 60,
152
+ tcp_user_timeout: nil,
153
+ frame_max: 131_072,
154
+ channel_max: 2047,
155
+ connect_timeout: CONNECT_TIMEOUT,
156
+ rpc_timeout: RPC_TIMEOUT,
157
+ auth_mechanism: nil,
158
+ connection_name: nil,
159
+ auto_recover: true,
160
+ recovery_attempts: nil,
161
+ recovery_interval: RECOVERY_INITIAL,
162
+ recovery_max_interval: RECOVERY_MAX,
163
+ recover_topology: true,
164
+ topology_recovery_filter: nil,
165
+ instrumenter: nil,
166
+ notifier: nil,
167
+ logger: Log.new
168
+ )
169
+ @addresses = self.class.address_list(host: host, port: port, hosts: hosts, addresses: addresses)
170
+ @host = @addresses.first[0]
171
+ @port = @addresses.first[1]
172
+ @hosts_shuffle_strategy = hosts_shuffle_strategy
173
+ @vhost = vhost
174
+ @username = username
175
+ @password = password
176
+ # Certificate material or a ready context means TLS, whatever tls: says.
177
+ @tls = tls || !(tls_context || tls_cert || tls_key || tls_ca_certificates).nil?
178
+ @tls_context = tls_context
179
+ @tls_cert = tls_cert
180
+ @tls_key = tls_key
181
+ @tls_ca_certificates = tls_ca_certificates
182
+ @verify_peer = verify_peer
183
+ @tls_min_version = tls_min_version
184
+ @heartbeat = heartbeat
185
+ @tcp_user_timeout = tcp_user_timeout
186
+ @frame_max = frame_max
187
+ @channel_max = channel_max
188
+ @connect_timeout = connect_timeout
189
+ @rpc_timeout = rpc_timeout
190
+ @auth_mechanism = auth_mechanism
191
+ @connection_name = connection_name
192
+ @auto_recover = auto_recover
193
+ @recovery_attempts = recovery_attempts # nil = unlimited
194
+ @recovery_interval = recovery_interval
195
+ @recovery_max_interval = recovery_max_interval
196
+ @recover_topology = recover_topology
197
+ @topology = TopologyRegistry.new(filter: topology_recovery_filter)
198
+ @logger = logger
199
+ @notifier = notifier || Notifier.new(logger: logger)
200
+ @notifier.subscribe { |name, payload| instrumenter.call(name, payload) } if instrumenter
201
+
202
+ @state = :closed
203
+ @frame_io = nil
204
+ @channels = {} # channel_id => Channel; a closed channel's id is free again
205
+ @channel_mutex = Async::Semaphore.new(1)
206
+ @negotiated_hb = nil
207
+ @negotiated_fm = nil
208
+ @negotiated_cmax = nil
209
+ @recovery_in_progress = false
210
+ @recovery_interrupted = false
211
+ @closed_by_user = false
212
+ @connecting = false
213
+ @open_condition = nil
214
+ @last_frame_at = nil
215
+ @heartbeat_task = nil
216
+ @channel0_task = nil # drains channel-0 queue; handles connection.blocked/unblocked
217
+ @recovery_task = nil
218
+ @recovery_wakeup = nil # Async::Condition to interrupt the retry sleep
219
+ @session_root_task = nil # top-level task; recovery is spawned here so it
220
+ # survives reader-task Cancel propagation
221
+ @on_blocked = nil
222
+ @on_unblocked = nil
223
+ @update_secret_condition = nil
224
+ @on_recovery_attempt = nil
225
+ @on_recovery = nil
226
+ @on_recovery_exhausted = nil
227
+ @on_connection_lost = nil
228
+ end
229
+
230
+ # Connect and complete AMQP handshake. Raises ConnectionTimeoutError if
231
+ # the handshake does not complete within +connect_timeout+ seconds.
232
+ # Raises NotOpenError if already connected.
233
+ def connect
234
+ raise NotOpenError, "Session is already connected" if open?
235
+ @connecting = true
236
+ @closed_by_user = false # a previous failed connect must not disable recovery
237
+ @session_root_task = Async::Task.current
238
+
239
+ last_error = nil
240
+ started_at = instrument_clock
241
+ shuffled_addresses.each do |target_host, target_port|
242
+ begin
243
+ Async::Task.current.with_timeout(@connect_timeout) do
244
+ raw_socket = open_socket(target_host, target_port)
245
+ @frame_io = build_frame_io(raw_socket)
246
+ @frame_io.start(spawn: method(:spawn_background))
247
+ handshake
248
+ start_heartbeat_task
249
+ start_channel0_monitor_task
250
+ @host = target_host
251
+ @port = target_port
252
+ @state = :open
253
+ @channel_ids = fresh_channel_ids
254
+ end
255
+ @connecting = false
256
+ instrument("connection.open") do
257
+ { host: @host, port: @port, vhost: @vhost, tls: @tls, heartbeat: @negotiated_hb,
258
+ frame_max: @negotiated_fm, channel_max: @negotiated_cmax,
259
+ duration: started_at ? instrument_elapsed(started_at) : nil }
260
+ end
261
+ return
262
+ rescue Async::TimeoutError => e
263
+ @frame_io&.stop rescue nil
264
+ @frame_io = nil
265
+ last_error = e
266
+ rescue SystemCallError, EOFError, IOError, SocketError => e
267
+ # TCP connect failed (refused, unreachable, timed out) or the peer
268
+ # dropped the connection mid-handshake (reset, EOF, shutdown).
269
+ @frame_io&.stop rescue nil
270
+ @frame_io = nil
271
+ last_error = e
272
+ rescue OpenSSL::SSL::SSLError => e
273
+ @frame_io&.stop rescue nil
274
+ @frame_io = nil
275
+ last_error = e
276
+ rescue AuthenticationError
277
+ # Credentials or vhost access refused: other nodes will say the same.
278
+ cleanup_after_failed_connect
279
+ raise
280
+ rescue ConnectionError, ChannelError => e
281
+ # The broker refused the handshake (e.g. 530 NOT_ALLOWED for a missing
282
+ # vhost, 320 CONNECTION_FORCED from a node in maintenance). Stop the
283
+ # reader/writer tasks on this socket and try the next address.
284
+ @frame_io&.stop rescue nil
285
+ @frame_io = nil
286
+ last_error = e
287
+ end
288
+ end
289
+
290
+ # All addresses exhausted
291
+ cleanup_after_failed_connect
292
+ tried = @addresses.map { |h, p| "#{h}:#{p}" }.join(", ")
293
+ case last_error
294
+ when Async::TimeoutError
295
+ raise ConnectionTimeoutError, "AMQP handshake did not complete within #{@connect_timeout}s (tried #{tried})"
296
+ when OpenSSL::SSL::SSLError
297
+ raise ConnectionTimeoutError, "TLS handshake failed (tried #{tried}) — #{last_error.message}"
298
+ when ConnectionError, ChannelError
299
+ raise last_error
300
+ else
301
+ raise ConnectionTimeoutError, "Could not connect to any host (tried #{tried}) — #{last_error&.message}"
302
+ end
303
+ end
304
+
305
+ def open?
306
+ @state == :open
307
+ end
308
+
309
+ def closed?
310
+ @state == :closed
311
+ end
312
+
313
+ # Internal. Run a long-lived background loop (channel dispatch, heartbeat,
314
+ # channel-0 monitor).
315
+ #
316
+ # These must not become children of whichever task happened to open the
317
+ # channel or the session: a short-lived caller that finishes, or is stopped,
318
+ # would silently take its dispatch loop with it while the channel still
319
+ # reported itself open. Parent them at the reactor instead, and mark them
320
+ # transient so they never hold the reactor open by themselves — #close
321
+ # stops them explicitly.
322
+ #
323
+ # @api private
324
+ def spawn_background(&block)
325
+ # Looked up per call, never memoised: a Session reused across two
326
+ # separate Sync/Async reactors would otherwise keep spawning onto the
327
+ # first, finished one.
328
+ Async::Task.current.root.async(transient: true, &block)
329
+ end
330
+
331
+ # Close the session gracefully. Also works if recovery is in progress.
332
+ def close
333
+ return if closed?
334
+ @closed_by_user = true
335
+ @recovery_wakeup&.signal rescue nil
336
+ # Unblock any fibers waiting on channel replies (e.g. wait_for inside
337
+ # reopen_after_recovery) so that @recovery_task.cancel below can actually
338
+ # terminate the task rather than leaving it stuck on an Async::Condition.
339
+ conn_error = ConnectionError.new(code: 0, text: "Session closed by user")
340
+ @channels.values.each { |ch| ch.mark_closed!(conn_error) rescue nil }
341
+ if open?
342
+ @state = :closing
343
+ @channel0_task&.cancel rescue nil
344
+ send_connection_close rescue nil
345
+ end
346
+ @heartbeat_task&.cancel rescue nil
347
+ @channel0_task&.cancel rescue nil
348
+ @recovery_task&.cancel rescue nil
349
+ @frame_io&.stop rescue nil
350
+ @state = :closed
351
+ instrument("connection.closed") { { reason: :user, host: @host, port: @port } }
352
+ end
353
+
354
+ # Open a new channel. Returns an AsyncRabbitMQ::Channel.
355
+ # +pool_size+ bounds concurrent consumer-handler fibers on the channel
356
+ # (Bunny-parity: default 1). basic_qos will auto-adjust it to prefetch_count.
357
+ def open_channel(pool_size: 1)
358
+ raise NotOpenError, "Session is not open" unless open?
359
+ channel = @channel_mutex.acquire do
360
+ channel_id = next_channel_id
361
+ Channel.new(channel_id, self, @frame_io, frame_max: @negotiated_fm || @frame_max, logger: @logger,
362
+ pool_size: pool_size, rpc_timeout: @rpc_timeout).tap { |ch| @channels[channel_id] = ch }
363
+ end
364
+ channel.open
365
+ channel
366
+ end
367
+
368
+ # Open a channel, yield it to the block, and ensure it is closed afterward.
369
+ def with_channel(pool_size: 1)
370
+ ch = open_channel(pool_size: pool_size)
371
+ begin
372
+ yield ch
373
+ ensure
374
+ ch.close rescue nil
375
+ end
376
+ end
377
+
378
+ # Rotate the secret this connection authenticated with, without
379
+ # reconnecting (connection.update-secret, RabbitMQ 3.8+). Used with the
380
+ # OAuth 2 auth backend to hand the broker a refreshed access token before
381
+ # the current one expires. The new secret is also used for reconnects.
382
+ # Raises ConnectionError if the broker refuses (e.g. an auth backend that
383
+ # does not support secret updates closes the connection).
384
+ def update_secret(new_secret, reason = "secret update")
385
+ raise NotOpenError, "Session is not open" unless open?
386
+
387
+ cond = @update_secret_condition = Async::Condition.new
388
+ @frame_io.write_frame(AMQ::Protocol::Connection::UpdateSecret.encode(new_secret, reason).encode)
389
+ result = Async::Task.current.with_timeout(@rpc_timeout || RPC_TIMEOUT) { cond.wait }
390
+ raise result if result.is_a?(Exception)
391
+
392
+ @password = new_secret
393
+ true
394
+ rescue Async::TimeoutError
395
+ raise RpcTimeoutError, "No reply to connection.update-secret within #{@rpc_timeout || RPC_TIMEOUT}s"
396
+ ensure
397
+ @update_secret_condition = nil
398
+ end
399
+
400
+ # Check whether a queue exists on the broker without creating it.
401
+ # Opens a temporary channel and performs a passive declare.
402
+ def queue_exists?(name)
403
+ with_channel { |ch| ch.queue(name, passive: true) }
404
+ true
405
+ rescue ChannelError
406
+ false
407
+ end
408
+
409
+ # Check whether an exchange exists on the broker without creating it.
410
+ # Opens a temporary channel and performs a passive declare.
411
+ def exchange_exists?(name)
412
+ with_channel { |ch| ch.exchange(name, passive: true) }
413
+ true
414
+ rescue ChannelError
415
+ false
416
+ end
417
+
418
+ # Register a callback invoked when the broker sends connection.blocked.
419
+ # The block receives the reason string from the broker.
420
+ def on_blocked(&block)
421
+ @on_blocked = block
422
+ end
423
+
424
+ # Register a callback invoked when the broker sends connection.unblocked.
425
+ def on_unblocked(&block)
426
+ @on_unblocked = block
427
+ end
428
+
429
+ # Register a callback invoked at the start of each recovery attempt.
430
+ # The block receives the attempt number (1-based).
431
+ def on_recovery_attempt(&block)
432
+ @on_recovery_attempt = block
433
+ end
434
+
435
+ # Register a callback invoked after recovery succeeds.
436
+ # The block receives the session.
437
+ def on_recovery(&block)
438
+ @on_recovery = block
439
+ end
440
+
441
+ # Subscribe to structured events from this session and its channels. The
442
+ # block receives (name, payload). +pattern+ is nil for every event, a
443
+ # String for one name or a prefix ending in a dot, or a Regexp. Returns a
444
+ # handle for notifier.unsubscribe. See Notifier::EVENTS for the taxonomy.
445
+ #
446
+ # session.on_event("message.") { |name, payload| statsd.increment(name) }
447
+ def on_event(pattern = nil, &block)
448
+ @notifier.subscribe(pattern, &block)
449
+ end
450
+
451
+ # Register a callback invoked when recovery attempts are exhausted.
452
+ # Only fires when recovery_attempts is set to a finite number.
453
+ # The block receives the session.
454
+ def on_recovery_exhausted(&block)
455
+ @on_recovery_exhausted = block
456
+ end
457
+
458
+ # Register a callback invoked when the connection is lost, before
459
+ # recovery starts (with auto_recover: false, after the channels have
460
+ # been closed). The block receives the session and the error.
461
+ def on_connection_lost(&block)
462
+ @on_connection_lost = block
463
+ end
464
+
465
+ # The channels open or recovering on this connection.
466
+ def channels
467
+ @channels.values
468
+ end
469
+
470
+ def channel_count
471
+ @channels.size
472
+ end
473
+
474
+ # Record a rotated secret for the next connect or reconnect without
475
+ # sending it: #update_secret does this itself on a live connection.
476
+ # For a session that is down while the secret rotates.
477
+ def store_secret(new_secret)
478
+ @password = new_secret
479
+ end
480
+
481
+ # Called by Channel when it closes itself.
482
+ def channel_closed(channel_id)
483
+ @channel_mutex.acquire do
484
+ @channels.delete(channel_id)
485
+ @channel_ids&.release(channel_id)
486
+ end
487
+ @frame_io&.unregister_channel(channel_id)
488
+ end
489
+
490
+ # Put a channel the broker closed back into the channel table and reopen
491
+ # it on the current connection under its original id (Channel#reopen).
492
+ def reopen_channel(channel, state: :open)
493
+ raise NotOpenError, "Session is not open" unless open?
494
+ @channel_mutex.acquire do
495
+ existing = @channels[channel.channel_id]
496
+ if existing && !existing.equal?(channel)
497
+ raise Error, "Channel id #{channel.channel_id} is in use by another channel"
498
+ end
499
+ @channel_ids ||= fresh_channel_ids
500
+ @channel_ids.reserve(channel.channel_id) unless existing
501
+ @channels[channel.channel_id] = channel
502
+ end
503
+ channel.reopen_on(@frame_io, state: state)
504
+ end
505
+
506
+ # Called by a channel when topology recovery re-declared a server-named
507
+ # queue under a new name: update the registry and every consumer of it.
508
+ def queue_renamed(old_name, new_name)
509
+ @logger.info("Server-named queue #{old_name} recovered as #{new_name}")
510
+ @topology.rename_queue(old_name, new_name)
511
+ @channels.values.each { |ch| ch.rename_consumer_queue(old_name, new_name) }
512
+ end
513
+
514
+ # --- Async::Pool resource interface ---
515
+
516
+ # Can this session be returned to the pool?
517
+ def reusable?
518
+ open? && !@recovery_in_progress
519
+ end
520
+
521
+ # Is the underlying socket alive?
522
+ def viable?
523
+ open?
524
+ end
525
+
526
+ # Max concurrent channels (negotiated with broker, or AMQP max 2047).
527
+ def concurrency
528
+ @negotiated_cmax || 2047
529
+ end
530
+
531
+ # --- Internal ---
532
+
533
+ # Called by FrameIO when a connection-level error triggers recovery.
534
+ def trigger_recovery(error, from: nil)
535
+ return if @closed_by_user
536
+ # A frame_io that a later connection attempt has already replaced is
537
+ # still draining its dead socket, and its errors are not the live
538
+ # connection's. Letting them through made a reconnect look like it had
539
+ # dropped again the moment it succeeded.
540
+ return if from && !@frame_io.equal?(from)
541
+
542
+ if @connecting
543
+ # The handshake is in flight in another fiber; hand it the IO error so
544
+ # it fails now (as a connect failure) rather than sitting in
545
+ # wait_channel0_method until the connect timeout expires.
546
+ q0 = @frame_io&.channel_queue(0)
547
+ q0&.push([:method, error]) rescue nil
548
+ return
549
+ end
550
+
551
+ instrument("connection.lost") do
552
+ { host: @host, port: @port, error: error.class.name, message: error.message,
553
+ recovering: @auto_recover }
554
+ end
555
+
556
+ unless @auto_recover
557
+ @state = :closed
558
+ conn_error = ConnectionError.new(code: 0, text: "Connection lost: #{error.message}")
559
+ @channels.values.each { |ch| ch.mark_closed!(conn_error) rescue nil }
560
+ connection_lost(error)
561
+ @frame_io&.stop rescue nil
562
+ return
563
+ end
564
+
565
+ @logger.warn("Connection lost (#{error.class}: #{error.message}). Starting recovery...")
566
+
567
+ if @recovery_in_progress
568
+ # Connection died again while a recovery attempt is in progress.
569
+ # Unblock any fiber stuck in wait_channel0_method or channel wait_for
570
+ # so that the current recover_loop iteration fails fast and retries.
571
+ recovery_error = ConnectionError.new(code: 0, text: "Connection lost during recovery")
572
+ @recovery_interrupted = true
573
+ q0 = @frame_io&.channel_queue(0)
574
+ q0&.push([:method, recovery_error]) rescue nil
575
+ @channels.each_value { |ch| ch.interrupt_wait!(recovery_error) rescue nil }
576
+ return
577
+ end
578
+
579
+ @recovery_in_progress = true
580
+ @state = :recovering
581
+
582
+ # Stop the channel-0 monitor so it doesn't race on the stale queue.
583
+ # Also unblock any fibers stuck in write_frame waiting on connection.blocked.
584
+ @channel0_task&.cancel rescue nil
585
+ @channel0_task = nil
586
+ @frame_io&.set_unblocked rescue nil
587
+
588
+ # Waits in flight raise ConnectionError instead of hanging; new operations
589
+ # on the channels park until they are reopened after reconnect.
590
+ conn_error = ConnectionError.new(code: 0, text: "Connection lost: #{error.message}")
591
+ @channels.values.each { |ch| ch.mark_recovering!(conn_error) rescue nil }
592
+ @update_secret_condition&.signal(conn_error)
593
+ # Tell the owner now, while the channels are still in the table: a
594
+ # Cluster in :drop mode gives them up here, before they are reopened.
595
+ connection_lost(error)
596
+
597
+ # Schedule recover_loop BEFORE stopping old frame_io. old_io.stop
598
+ # cancels reader/writer tasks; if we ARE the reader task, cancel raises
599
+ # Async::Cancel (< Exception) which bypasses `rescue nil` and propagates,
600
+ # so anything after stop might not run.
601
+ # Parent recovery at the reactor, not at Async::Task.current — which here
602
+ # is usually the reader task, whose Cancel would propagate into a child
603
+ # recovery task — and not at the task that called connect either, since
604
+ # that one may be short-lived while the session outlives it.
605
+ @recovery_task = spawn_background { recover_loop }
606
+
607
+ # Stop the old frame_io: closes the dead socket, pushes nil to channel
608
+ # queues (unblocking wait_channel0_method), and cancels writer task.
609
+ # Reader task cancel may raise Async::Cancel — recovery is already scheduled.
610
+ old_io = @frame_io
611
+ @frame_io = nil
612
+ old_io&.stop rescue nil
613
+ end
614
+
615
+ def frame_max
616
+ @negotiated_fm || @frame_max
617
+ end
618
+
619
+ private
620
+
621
+ # Hand the connection-lost callback its error without letting a failure
622
+ # in it stop recovery.
623
+ def connection_lost(error)
624
+ @on_connection_lost&.call(self, error)
625
+ rescue => e
626
+ @logger.error("on_connection_lost callback failed: #{e.class}: #{e.message}")
627
+ end
628
+
629
+ # Return the address list in the order they should be tried for this
630
+ # connect/recovery cycle.
631
+ def shuffled_addresses
632
+ case @hosts_shuffle_strategy
633
+ when :shuffle then @addresses.shuffle
634
+ when :none then @addresses.dup
635
+ when Proc then @hosts_shuffle_strategy.call(@addresses)
636
+ else @addresses.shuffle
637
+ end
638
+ end
639
+
640
+ # Stop any in-flight frame_io / recovery tasks spawned during a failed connect.
641
+ def cleanup_after_failed_connect
642
+ @closed_by_user = true
643
+ @connecting = false
644
+ @recovery_task&.cancel rescue nil
645
+ @recovery_task = nil
646
+ @heartbeat_task&.cancel rescue nil
647
+ @channel0_task&.cancel rescue nil
648
+ @frame_io&.stop rescue nil
649
+ @frame_io = nil
650
+ @state = :closed
651
+ end
652
+
653
+ def open_socket(target_host = @host, target_port = @port)
654
+ if @tls
655
+ require "openssl"
656
+ ctx = @tls_context || build_tls_context
657
+ raw = TCPSocket.new(target_host, target_port)
658
+ # SSLSocket does not close the socket it wraps when the handshake fails,
659
+ # and a failed connect has no frame_io to stop, so without this a broker
660
+ # with a bad certificate leaks one descriptor per recovery attempt.
661
+ handshaked = false
662
+ begin
663
+ ssl = OpenSSL::SSL::SSLSocket.new(raw, ctx)
664
+ # Without this, closing the SSL socket sends close_notify but leaves
665
+ # the TCP socket open until the GC finalises it, so every amqps
666
+ # reconnect strands a connection (Amazon MQ, for example).
667
+ ssl.sync_close = true
668
+ ssl.hostname = target_host
669
+ ssl.connect
670
+ handshaked = true
671
+ apply_socket_timeouts(raw)
672
+ ssl
673
+ ensure
674
+ raw.close unless handshaked
675
+ end
676
+ else
677
+ TCPSocket.new(target_host, target_port).tap { |sock| apply_socket_timeouts(sock) }
678
+ end
679
+ end
680
+
681
+ # Bound how long the kernel will retransmit unacknowledged data before it
682
+ # gives up on the socket. Without it a peer that disappears mid-write (a
683
+ # network partition with a full send buffer) is only noticed when the TCP
684
+ # retransmit timer expires, which on Linux is around 15 minutes — far
685
+ # longer than the heartbeat timeout we promise callers.
686
+ #
687
+ # TCP_USER_TIMEOUT is Linux-only; elsewhere the heartbeat task's own write
688
+ # timeout is the backstop, so a missing constant is not an error.
689
+ def apply_socket_timeouts(sock)
690
+ millis = @tcp_user_timeout
691
+ millis ||= (@heartbeat.to_i > 0 ? @heartbeat.to_i * 2 * 1000 : nil)
692
+ return unless millis && millis > 0
693
+ return unless Socket.const_defined?(:TCP_USER_TIMEOUT)
694
+
695
+ sock.setsockopt(Socket::IPPROTO_TCP, Socket::TCP_USER_TIMEOUT, millis)
696
+ rescue StandardError => e
697
+ # Unsupported on this platform or socket type; the heartbeat still covers us.
698
+ @logger.debug("Could not set TCP_USER_TIMEOUT: #{e.class}: #{e.message}")
699
+ end
700
+
701
+ def build_tls_context
702
+ TLS.context(cert: @tls_cert, key: @tls_key, ca_certificates: @tls_ca_certificates,
703
+ verify_peer: @verify_peer, min_version: @tls_min_version)
704
+ end
705
+
706
+ def build_frame_io(raw_socket)
707
+ io = FrameIO.new(raw_socket, logger: @logger)
708
+ # Register channel 0 for connection-level frames
709
+ io.register_channel(0)
710
+
711
+ # Override trigger_recovery to delegate to Session
712
+ session = self
713
+ io.define_singleton_method(:trigger_recovery) { |error| session.trigger_recovery(error, from: io) }
714
+
715
+ # Update heartbeat timestamp on every received frame
716
+ io.on_frame = -> { @last_frame_at = Process.clock_gettime(Process::CLOCK_MONOTONIC) }
717
+ io
718
+ end
719
+
720
+ def handshake
721
+ # Send AMQP protocol header directly on the socket
722
+ @frame_io.instance_variable_get(:@socket).write(PROTOCOL_HEADER)
723
+
724
+ # connection.start — negotiate SASL mechanism
725
+ start = wait_channel0_method(AMQ::Protocol::Connection::Start)
726
+ sasl = SASL.negotiate(
727
+ start.mechanisms,
728
+ preferred: @auth_mechanism,
729
+ username: @username,
730
+ password: @password
731
+ )
732
+ send_connection_start_ok(sasl)
733
+
734
+ # connection.tune (broker may send connection.secure challenges first)
735
+ msg = wait_channel0_method(AMQ::Protocol::Connection::Tune, AMQ::Protocol::Connection::Secure)
736
+ while msg.is_a?(AMQ::Protocol::Connection::Secure)
737
+ @logger.debug("Received connection.secure SASL challenge for #{sasl.mechanism_name}")
738
+ @frame_io.write_frame(AMQ::Protocol::Connection::SecureOk.encode(sasl.challenge_response(msg.challenge)).encode)
739
+ msg = wait_channel0_method(AMQ::Protocol::Connection::Tune, AMQ::Protocol::Connection::Secure)
740
+ end
741
+ @negotiated_hb = negotiate_heartbeat(msg.heartbeat)
742
+ @negotiated_fm = negotiate_frame_max(msg.frame_max)
743
+ @negotiated_cmax = negotiate_channel_max(msg.channel_max)
744
+ send_connection_tune_ok
745
+
746
+ # connection.open
747
+ send_connection_open
748
+ wait_channel0_method(AMQ::Protocol::Connection::OpenOk)
749
+ end
750
+
751
+ def wait_channel0_method(*expected_classes)
752
+ queue = @frame_io.channel_queue(0)
753
+ loop do
754
+ msg = queue.pop
755
+ # nil sentinel — pushed by frame_io.stop or interrupt during recovery
756
+ raise ConnectionError, "Connection closed while waiting for #{expected_classes.join(', ')}" if msg.nil?
757
+ next unless msg[0] == :method
758
+ method = msg[1]
759
+ # An exception pushed directly by trigger_recovery to unblock this wait
760
+ # (ConnectionError during recovery, or the raw IO error during connect).
761
+ raise method if method.is_a?(Exception)
762
+ if expected_classes.any? { |c| method.is_a?(c) }
763
+ return method
764
+ elsif method.is_a?(AMQ::Protocol::Connection::Close)
765
+ code = method.reply_code
766
+ text = method.reply_text
767
+ # connection.close is always connection-level. 403 ACCESS_REFUSED here
768
+ # means the credentials or the vhost access were rejected.
769
+ if code == 403
770
+ raise AuthenticationError.new(code: code, text: text)
771
+ else
772
+ raise ConnectionError.new(code: code, text: text)
773
+ end
774
+ end
775
+ end
776
+ end
777
+
778
+ def send_connection_start_ok(sasl)
779
+ props = {
780
+ "product" => "async-rabbitmq",
781
+ "version" => AsyncRabbitMQ::VERSION,
782
+ "platform" => "Ruby #{RUBY_VERSION}",
783
+ "information" => "https://github.com/womblep/async-rabbitmq",
784
+ # Extensions this client understands. authentication_failure_close makes
785
+ # RabbitMQ answer bad credentials with connection.close 403 instead of
786
+ # silently dropping the TCP connection (https://www.rabbitmq.com/docs/auth-notification).
787
+ "capabilities" => {
788
+ "publisher_confirms" => true,
789
+ "consumer_cancel_notify" => true,
790
+ "exchange_exchange_bindings" => true,
791
+ "basic.nack" => true,
792
+ "connection.blocked" => true,
793
+ "authentication_failure_close" => true,
794
+ },
795
+ }
796
+ props["connection_name"] = @connection_name if @connection_name
797
+ @frame_io.write_frame(
798
+ AMQ::Protocol::Connection::StartOk.encode(
799
+ props,
800
+ sasl.mechanism_name,
801
+ sasl.initial_response,
802
+ "en_US"
803
+ ).encode
804
+ )
805
+ end
806
+
807
+ def send_connection_tune_ok
808
+ @frame_io.write_frame(
809
+ AMQ::Protocol::Connection::TuneOk.encode(
810
+ @negotiated_cmax,
811
+ @negotiated_fm,
812
+ @negotiated_hb
813
+ ).encode
814
+ )
815
+ end
816
+
817
+ def send_connection_open
818
+ @frame_io.write_frame(
819
+ AMQ::Protocol::Connection::Open.encode(@vhost).encode
820
+ )
821
+ end
822
+
823
+ def send_connection_close
824
+ # Bound the whole handshake, write included: the broker may never answer,
825
+ # and the write queue may be full behind a stalled socket.
826
+ Async::Task.current.with_timeout(5) do
827
+ @frame_io.write_frame(
828
+ AMQ::Protocol::Connection::Close.encode(200, "Goodbye", 0, 0).encode
829
+ )
830
+ wait_channel0_method(AMQ::Protocol::Connection::CloseOk)
831
+ end
832
+ rescue Async::TimeoutError, ConnectionError, ChannelError, IOError => e
833
+ # Broker didn't answer in time — proceed with the forced close below, which
834
+ # takes the rest of the write queue with it. basic_publish only queues
835
+ # frames, so without confirms this is the publisher's only hint.
836
+ unwritten = @frame_io&.pending_writes.to_i
837
+ return unless unwritten.positive?
838
+
839
+ @logger.warn("Close discarded #{unwritten} queued write(s) (#{e.class})")
840
+ end
841
+
842
+ def negotiate_heartbeat(broker_hb)
843
+ return @heartbeat if broker_hb == 0
844
+ return broker_hb if @heartbeat == 0
845
+ [@heartbeat, broker_hb].min
846
+ end
847
+
848
+ def negotiate_frame_max(broker_fm)
849
+ return @frame_max if broker_fm == 0
850
+ [@frame_max, broker_fm].min
851
+ end
852
+
853
+ # 0 means "no limit" on either side; otherwise the lower value wins.
854
+ def negotiate_channel_max(broker_cmax)
855
+ return (@channel_max == 0 ? 2047 : @channel_max) if broker_cmax == 0
856
+ return broker_cmax if @channel_max == 0
857
+ [@channel_max, broker_cmax].min
858
+ end
859
+
860
+ # The negotiated value T is the heartbeat *timeout*. RabbitMQ and the
861
+ # reference clients send a heartbeat every T/2 and treat the peer as dead
862
+ # after roughly two missed heartbeats, so we send at T/2 as well and
863
+ # declare the broker dead when nothing has arrived for 2×T.
864
+ def start_heartbeat_task
865
+ timeout = @negotiated_hb || 60
866
+ return if timeout == 0
867
+ interval = timeout / 2.0
868
+ dead_after = timeout * 2
869
+ # Seed the timestamp now; the on_frame callback will keep it fresh.
870
+ @last_frame_at = Process.clock_gettime(Process::CLOCK_MONOTONIC)
871
+
872
+ @heartbeat_task = spawn_background do
873
+ loop do
874
+ sleep interval
875
+ break unless open?
876
+
877
+ # Liveness first. The write below needs the socket lock, and after a
878
+ # partition the writer can be parked inside @socket.write holding it
879
+ # with a full send buffer. Checking afterwards meant this task blocked
880
+ # with the rest and nothing declared the peer dead until the kernel
881
+ # gave up retransmitting — around 15 minutes on Linux.
882
+ last = @last_frame_at
883
+ if last
884
+ elapsed = Process.clock_gettime(Process::CLOCK_MONOTONIC) - last
885
+ if elapsed > dead_after
886
+ trigger_recovery(HeartbeatTimeoutError.new("No frame received in #{elapsed.round(1)}s (timeout #{dead_after}s)"))
887
+ break
888
+ end
889
+ end
890
+
891
+ # Bounded so a socket that never drains cannot park this task either.
892
+ # Bounded by the full dead-peer window, not by the send interval: the
893
+ # heartbeat needs the socket lock, and a large batch going out over a
894
+ # slow link can legitimately hold it for longer than one interval.
895
+ begin
896
+ Async::Task.current.with_timeout(dead_after) { @frame_io.write_heartbeat }
897
+ instrument("heartbeat.sent") { { interval: interval } }
898
+ rescue Async::TimeoutError
899
+ trigger_recovery(HeartbeatTimeoutError.new("Heartbeat write blocked for #{dead_after}s — peer is not reading"))
900
+ break
901
+ end
902
+ end
903
+ end
904
+ end
905
+
906
+ # The lowest channel id not in use, so ids come back when channels close.
907
+ # A bare counter ran out at 65535 (the channel field is 16 bits; 65536
908
+ # encodes as channel 0 and the broker drops the connection) and let the
909
+ # process open more channels than channel_max, which the broker also
910
+ # answers by closing the connection. Called with @channel_mutex held.
911
+ def next_channel_id
912
+ @channel_ids ||= fresh_channel_ids
913
+ @channel_ids.allocate or
914
+ raise ChannelLimitError, "channel_max #{@channel_ids.limit} reached: #{@channels.size} channels open on this connection"
915
+ end
916
+
917
+ # The negotiated channel_max; 0 means "no limit", which the protocol caps
918
+ # at 65535 through the width of the frame header's channel field.
919
+ def channel_limit
920
+ max = @negotiated_cmax || @channel_max
921
+ max.nil? || max.zero? ? 65_535 : max
922
+ end
923
+
924
+ # A new allocator sized by the current negotiation, with the ids of the
925
+ # channels we already have marked as taken. Built lazily and again after
926
+ # each handshake, since a reconnect may negotiate a different channel_max
927
+ # while the channels keep their ids.
928
+ def fresh_channel_ids
929
+ ChannelIdAllocator.new(channel_limit).tap do |ids|
930
+ @channels.each_key { |id| ids.reserve(id) }
931
+ end
932
+ end
933
+
934
+ def recover_loop
935
+ delay = @recovery_interval
936
+ attempts = 0
937
+ recovery_started_at = instrument_clock
938
+
939
+ loop do
940
+ break if @closed_by_user
941
+
942
+ attempts += 1
943
+
944
+ # Check retry limit (nil = unlimited)
945
+ if @recovery_attempts && attempts > @recovery_attempts
946
+ @logger.warn("Recovery exhausted after #{@recovery_attempts} attempt(s)")
947
+ @state = :closed
948
+ @recovery_in_progress = false
949
+ exhausted = ConnectionError.new(code: 0, text: "Recovery exhausted after #{@recovery_attempts} attempt(s)")
950
+ @channels.values.each { |ch| ch.mark_closed!(exhausted) rescue nil }
951
+ instrument("recovery.exhausted") { { attempts: @recovery_attempts, reason: :attempts_exceeded } }
952
+ @on_recovery_exhausted&.call(self)
953
+ return
954
+ end
955
+
956
+ jitter = delay * RECOVERY_JITTER * (rand * 2 - 1)
957
+ wait_secs = delay + jitter
958
+
959
+ # Sleep until the delay expires OR session.close signals @recovery_wakeup.
960
+ @recovery_wakeup = Async::Condition.new
961
+ Async::Task.current.with_timeout(wait_secs) do
962
+ @recovery_wakeup.wait
963
+ end rescue nil # TimeoutError is normal; Condition#signal raises nothing
964
+ @recovery_wakeup = nil
965
+
966
+ break if @closed_by_user
967
+
968
+ @logger.info("Recovery attempt #{attempts} (delay was #{delay.round(1)}s)...")
969
+ instrument("recovery.attempt") { { attempt: attempts, delay: wait_secs } }
970
+ @on_recovery_attempt&.call(attempts)
971
+
972
+ connected = false
973
+ shuffled_addresses.each do |target_host, target_port|
974
+ break if @closed_by_user
975
+ begin
976
+ # Wrap the entire reconnect in a timeout so a dead socket doesn't hang forever.
977
+ Async::Task.current.with_timeout(@connect_timeout) do
978
+ raw_socket = open_socket(target_host, target_port)
979
+ @frame_io = build_frame_io(raw_socket)
980
+ @frame_io.start(spawn: method(:spawn_background))
981
+ handshake
982
+ end
983
+ @host = target_host
984
+ @port = target_port
985
+ connected = true
986
+ break
987
+ rescue AuthenticationError => e
988
+ # The credentials no longer work; retrying cannot help.
989
+ @frame_io&.stop rescue nil
990
+ @frame_io = nil
991
+ @logger.error("Recovery abandoned: #{e.message}")
992
+ @state = :closed
993
+ @recovery_in_progress = false
994
+ @channels.values.each { |ch| ch.mark_closed!(e) rescue nil }
995
+ instrument("recovery.exhausted") { { attempts: attempts, reason: :authentication_failed } }
996
+ @on_recovery_exhausted&.call(self)
997
+ @recovery_task = nil
998
+ return
999
+ rescue => e
1000
+ @frame_io&.stop rescue nil
1001
+ @logger.debug("Recovery: #{target_host}:#{target_port} failed — #{e.class}: #{e.message}")
1002
+ end
1003
+ end
1004
+
1005
+ if connected
1006
+ # Only drops from here on belong to this connection: the old socket
1007
+ # goes on failing throughout the backoff, and those errors must not
1008
+ # make the reconnect we just made look like it had dropped too.
1009
+ @recovery_interrupted = false
1010
+ @heartbeat_task&.cancel rescue nil
1011
+ start_heartbeat_task
1012
+ start_channel0_monitor_task
1013
+ @state = :open
1014
+ @channel_ids = fresh_channel_ids
1015
+ # NOTE: keep @recovery_task non-nil until reopen_channels completes so
1016
+ # that session.close can still cancel this task (and therefore interrupt
1017
+ # any wait_for calls inside reopen_after_recovery) if the user closes
1018
+ # the session while channels are being reopened.
1019
+ @logger.info("Recovery successful after #{attempts} attempt(s)")
1020
+
1021
+ # Re-open channels and re-register consumers. @recovery_in_progress
1022
+ # stays true across this: clearing it first leaves a window where a
1023
+ # second drop starts a competing recovery while this one is still
1024
+ # failing channels through mark_closed!, which loses those channels
1025
+ # and their consumers for good.
1026
+ unless reopen_channels
1027
+ # Dropped again mid-reopen. Channels are still :recovering, so go
1028
+ # round again and reopen them on the next connection — but tear
1029
+ # this half-built connection down first. Leaving it up reported the
1030
+ # session as open (so a Cluster would place new channels on a dead
1031
+ # node), held the socket, and leaked a channel-0 monitor per flap.
1032
+ @logger.warn("Connection lost while reopening channels; retrying recovery")
1033
+ @state = :recovering
1034
+ @heartbeat_task&.cancel rescue nil
1035
+ @heartbeat_task = nil
1036
+ @channel0_task&.cancel rescue nil
1037
+ @channel0_task = nil
1038
+ @frame_io&.stop rescue nil
1039
+ delay = [delay * 2, @recovery_max_interval].min
1040
+ next
1041
+ end
1042
+
1043
+ @recovery_in_progress = false
1044
+ instrument("recovery.succeeded") do
1045
+ { attempts: attempts, host: @host, port: @port, channels: @channels.size,
1046
+ duration: recovery_started_at ? instrument_elapsed(recovery_started_at) : nil }
1047
+ end
1048
+ @on_recovery&.call(self)
1049
+ @recovery_task = nil
1050
+ return
1051
+ else
1052
+ @logger.warn("Recovery attempt #{attempts} failed: no reachable host")
1053
+ delay = [delay * 2, @recovery_max_interval].min
1054
+ break if @closed_by_user
1055
+ end
1056
+ end
1057
+ ensure
1058
+ # Make sure state is consistent if we exit for any reason
1059
+ @recovery_in_progress = false if @closed_by_user
1060
+ end
1061
+
1062
+ # After reconnect: reopen every channel, replay the recorded topology
1063
+ # (session-wide, in dependency order), then release parked callers and
1064
+ # re-register consumers.
1065
+ # Returns false if the connection dropped again while channels were being
1066
+ # reopened. Channels are then left :recovering rather than closed, so the
1067
+ # next pass through recover_loop can reopen them; only a failure that is
1068
+ # this channel's own (a broker rejection) closes it for good.
1069
+ def reopen_channels
1070
+ reopened = []
1071
+
1072
+ @channels.values.each do |channel|
1073
+ begin
1074
+ channel.reopen_on(@frame_io, state: :recovering)
1075
+ reopened << channel
1076
+ rescue => e
1077
+ return false if @recovery_interrupted
1078
+
1079
+ # Fail the channel loudly rather than leave its parked callers hanging.
1080
+ @logger.error("Channel #{channel.channel_id} could not be reopened after recovery: #{e.class}: #{e.message}")
1081
+ channel.mark_closed!(ChannelError.new("Channel could not be reopened after recovery: #{e.message}",
1082
+ channel_id: channel.channel_id))
1083
+ channel_closed(channel.channel_id)
1084
+ end
1085
+ end
1086
+
1087
+ recover_topology_on(reopened) if @recover_topology && !@topology.empty?
1088
+ return false if @recovery_interrupted
1089
+
1090
+ reopened.each { |channel| channel.finish_recovery! rescue nil }
1091
+ true
1092
+ end
1093
+
1094
+ # Re-declare exchanges, then queues, then bindings. Each entity is replayed
1095
+ # on the channel that declared it if that channel is still open, otherwise
1096
+ # on a temporary channel. One entity failing does not stop the others.
1097
+ def recover_topology_on(channels)
1098
+ by_id = channels.to_h { |ch| [ch.channel_id, ch] }
1099
+ temp = nil
1100
+ on = ->(channel_id) { by_id[channel_id] || (temp ||= open_channel) }
1101
+
1102
+ @topology.exchanges.each { |x| on.call(x.channel_id).recover_exchange(x) }
1103
+ @topology.queues.each { |q| on.call(q.channel_id).recover_queue(q) }
1104
+ @topology.queue_bindings.each { |b| on.call(b.channel_id).recover_queue_binding(b) }
1105
+ @topology.exchange_bindings.each { |b| on.call(b.channel_id).recover_exchange_binding(b) }
1106
+ rescue => e
1107
+ @logger.error("Topology recovery aborted: #{e.class}: #{e.message}")
1108
+ ensure
1109
+ temp&.close rescue nil
1110
+ end
1111
+
1112
+
1113
+ # Dedicated long-lived task that drains the channel-0 queue after the
1114
+ # AMQP handshake completes. Handles connection.blocked / connection.unblocked
1115
+ # by delegating to FrameIO's blocked-state gate so write_frame yields
1116
+ # automatically when the broker is resource-constrained.
1117
+ def start_channel0_monitor_task
1118
+ # Cancel any predecessor: a recovery that has to retry starts this again,
1119
+ # and the old one would otherwise sit on a dead queue for good.
1120
+ @channel0_task&.cancel rescue nil
1121
+ @channel0_task = spawn_background { channel0_monitor_loop }
1122
+ end
1123
+
1124
+ def channel0_monitor_loop
1125
+ queue = @frame_io.channel_queue(0)
1126
+ loop do
1127
+ msg = queue.pop
1128
+ break if msg.nil?
1129
+ next unless msg[0] == :method
1130
+ method = msg[1]
1131
+ case method
1132
+ when AMQ::Protocol::Connection::Blocked
1133
+ @frame_io&.set_blocked(method.reason)
1134
+ instrument("connection.blocked") { { reason: method.reason } }
1135
+ @on_blocked&.call(method.reason)
1136
+ when AMQ::Protocol::Connection::Unblocked
1137
+ @frame_io&.set_unblocked
1138
+ instrument("connection.unblocked") { {} }
1139
+ @on_unblocked&.call
1140
+ when AMQ::Protocol::Connection::UpdateSecretOk
1141
+ @update_secret_condition&.signal(method)
1142
+ when AMQ::Protocol::Connection::Close
1143
+ # FrameIO has already answered with close-ok and triggered recovery;
1144
+ # fail a pending update_secret with the broker's reason.
1145
+ @update_secret_condition&.signal(ConnectionError.new(code: method.reply_code, text: method.reply_text))
1146
+ end
1147
+ # Other channel-0 methods during normal operation are intentionally
1148
+ # ignored; the handshake uses wait_channel0_method, not this loop.
1149
+ end
1150
+ rescue => e
1151
+ @logger.debug("Channel-0 monitor exited: #{e.class}: #{e.message}")
1152
+ end
1153
+ end
1154
+ end