async-rabbitmq 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG.md +268 -0
- data/LICENSE +21 -0
- data/README.md +594 -0
- data/exe/async-rabbitmq +12 -0
- data/lib/async_rabbitmq/channel.rb +1422 -0
- data/lib/async_rabbitmq/channel_id_allocator.rb +57 -0
- data/lib/async_rabbitmq/cli.rb +300 -0
- data/lib/async_rabbitmq/cluster.rb +433 -0
- data/lib/async_rabbitmq/errors.rb +113 -0
- data/lib/async_rabbitmq/exchange.rb +70 -0
- data/lib/async_rabbitmq/frame_io.rb +309 -0
- data/lib/async_rabbitmq/log.rb +52 -0
- data/lib/async_rabbitmq/notifier.rb +122 -0
- data/lib/async_rabbitmq/pool.rb +43 -0
- data/lib/async_rabbitmq/queue.rb +88 -0
- data/lib/async_rabbitmq/sasl.rb +123 -0
- data/lib/async_rabbitmq/session.rb +1154 -0
- data/lib/async_rabbitmq/telemetry/open_telemetry.rb +211 -0
- data/lib/async_rabbitmq/tls.rb +82 -0
- data/lib/async_rabbitmq/topology_registry.rb +210 -0
- data/lib/async_rabbitmq/version.rb +3 -0
- data/lib/async_rabbitmq/versioned_delivery_tag.rb +62 -0
- data/lib/async_rabbitmq.rb +18 -0
- metadata +180 -0
|
@@ -0,0 +1,1154 @@
|
|
|
1
|
+
require "async"
|
|
2
|
+
require "async/condition"
|
|
3
|
+
require "async/semaphore"
|
|
4
|
+
require "socket"
|
|
5
|
+
require "openssl"
|
|
6
|
+
require "uri"
|
|
7
|
+
require "amq/uri"
|
|
8
|
+
require_relative "errors"
|
|
9
|
+
require_relative "frame_io"
|
|
10
|
+
require_relative "channel"
|
|
11
|
+
require_relative "sasl"
|
|
12
|
+
|
|
13
|
+
module AsyncRabbitMQ
|
|
14
|
+
# Represents one AMQP connection to a RabbitMQ broker.
|
|
15
|
+
#
|
|
16
|
+
# Recovery: exponential backoff with configurable interval, max interval,
|
|
17
|
+
# and retry limit. Re-registers consumers after reconnect.
|
|
18
|
+
#
|
|
19
|
+
# Pool interface: implements reusable?, viable?, concurrency, close for Async::Pool.
|
|
20
|
+
class Session
|
|
21
|
+
include Instrumented
|
|
22
|
+
|
|
23
|
+
CONNECT_TIMEOUT = 30 # seconds for AMQP handshake
|
|
24
|
+
RPC_TIMEOUT = 15 # seconds to wait for a synchronous channel reply (nil = forever)
|
|
25
|
+
RECOVERY_INITIAL = 1.0 # seconds
|
|
26
|
+
RECOVERY_MAX = 30.0 # seconds
|
|
27
|
+
RECOVERY_JITTER = 0.25 # ±25%
|
|
28
|
+
PROTOCOL_HEADER = "AMQP\x00\x00\x09\x01".b.freeze
|
|
29
|
+
|
|
30
|
+
attr_reader :host, :port, :vhost, :username, :addresses
|
|
31
|
+
|
|
32
|
+
# The TopologyRegistry of exchanges, queues and bindings declared through
|
|
33
|
+
# this session, replayed after a reconnect (see recover_topology:).
|
|
34
|
+
attr_reader :topology
|
|
35
|
+
|
|
36
|
+
# Structured events for metrics and tracing; see Notifier::EVENTS.
|
|
37
|
+
attr_reader :notifier
|
|
38
|
+
|
|
39
|
+
# Build a Session from one or more AMQP URI strings.
|
|
40
|
+
#
|
|
41
|
+
# Session.from_uri("amqp://user:pass@rabbit:5672/myvhost")
|
|
42
|
+
# Session.from_uri("amqps://rabbit/myvhost", tls_context: ctx)
|
|
43
|
+
# Session.from_uri("amqp://rabbit1:5672/vh", "amqp://rabbit2:5672/vh")
|
|
44
|
+
#
|
|
45
|
+
# When multiple URIs are given, credentials, vhost, and TLS settings are
|
|
46
|
+
# taken from the first URI. Each URI contributes a host:port pair to the
|
|
47
|
+
# +addresses+ list used for failover.
|
|
48
|
+
#
|
|
49
|
+
# Keyword arguments override anything parsed from the URI(s).
|
|
50
|
+
#
|
|
51
|
+
# The standard RabbitMQ URI query parameters are honoured: +heartbeat+,
|
|
52
|
+
# +connection_timeout+, +channel_max+, +auth_mechanism+ and, for amqps://,
|
|
53
|
+
# +verify+, +cacertfile+, +certfile+ and +keyfile+ (which build a
|
|
54
|
+
# +tls_context+ unless one is passed explicitly).
|
|
55
|
+
def self.from_uri(*uri_strings, **kwargs)
|
|
56
|
+
new(**options_from_uris(*uri_strings, **kwargs))
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
# The Session.new keyword arguments for one or more URIs; see from_uri.
|
|
60
|
+
# Cluster.from_uri builds its nodes from the same list.
|
|
61
|
+
def self.options_from_uris(*uri_strings, **kwargs)
|
|
62
|
+
raise ArgumentError, "at least one URI string is required" if uri_strings.empty?
|
|
63
|
+
|
|
64
|
+
first_opts = uri_options(uri_strings.first)
|
|
65
|
+
address_list = uri_strings.map do |u|
|
|
66
|
+
parsed = uri_options(u)
|
|
67
|
+
"#{parsed[:host] || 'localhost'}:#{parsed[:port]}"
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# Connection-level settings come from the first URI; addresses from all.
|
|
71
|
+
opts = first_opts.except(:host, :port)
|
|
72
|
+
opts[:addresses] = address_list
|
|
73
|
+
opts.merge(kwargs)
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
# Translate an amqp:// or amqps:// URI into Session.new keyword arguments.
|
|
77
|
+
#
|
|
78
|
+
# Parsing is delegated to AMQ::URI from amq-protocol, which percent-decodes
|
|
79
|
+
# the credentials and vhost (so "pa+ss" stays "pa+ss") and validates the
|
|
80
|
+
# scheme, the single-segment vhost path and the TLS-only query parameters.
|
|
81
|
+
def self.uri_options(uri_string)
|
|
82
|
+
parsed = AMQ::URI.parse(uri_string)
|
|
83
|
+
query = ::URI.parse(uri_string).query
|
|
84
|
+
params = query ? ::URI.decode_www_form(query).to_h : {}
|
|
85
|
+
|
|
86
|
+
opts = { tls: !!parsed[:ssl], port: parsed[:port] }
|
|
87
|
+
opts[:host] = parsed[:host] if parsed[:host]
|
|
88
|
+
opts[:username] = parsed[:user] if parsed[:user]
|
|
89
|
+
opts[:password] = parsed[:pass] if parsed[:pass]
|
|
90
|
+
# "amqp://host/" yields an empty vhost; treat it as the default "/" rather
|
|
91
|
+
# than the (rarely intended) vhost literally named "".
|
|
92
|
+
opts[:vhost] = parsed[:vhost].empty? ? "/" : parsed[:vhost] if parsed.key?(:vhost)
|
|
93
|
+
|
|
94
|
+
opts[:heartbeat] = Integer(params["heartbeat"]) if params.key?("heartbeat")
|
|
95
|
+
opts[:connect_timeout] = Integer(params["connection_timeout"]) if params.key?("connection_timeout")
|
|
96
|
+
opts[:channel_max] = Integer(params["channel_max"]) if params.key?("channel_max")
|
|
97
|
+
opts[:auth_mechanism] = params["auth_mechanism"] if params.key?("auth_mechanism")
|
|
98
|
+
|
|
99
|
+
if opts[:tls] && %w[verify cacertfile certfile keyfile].any? { |k| params.key?(k) }
|
|
100
|
+
opts[:tls_context] = tls_context_from_uri(parsed, params)
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
opts
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
# Build an OpenSSL context from the TLS query parameters of an amqps:// URI,
|
|
107
|
+
# the same way the tls_cert:/tls_key:/tls_ca_certificates: options do.
|
|
108
|
+
# Peer verification stays on unless the URI says verify=false explicitly.
|
|
109
|
+
def self.tls_context_from_uri(parsed, params)
|
|
110
|
+
verify = params.key?("verify") ? params["verify"] != "false" : true
|
|
111
|
+
TLS.context(cert: parsed[:certfile], key: parsed[:keyfile], ca_certificates: parsed[:cacertfile],
|
|
112
|
+
verify_peer: verify)
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
private_class_method :uri_options, :tls_context_from_uri
|
|
116
|
+
|
|
117
|
+
# The canonical [host, port] list from the various input forms.
|
|
118
|
+
#
|
|
119
|
+
# addresses: ["rabbit1:5672", "rabbit2:5673"] -> [["rabbit1",5672], ["rabbit2",5673]]
|
|
120
|
+
# hosts: ["rabbit1", "rabbit2"], port: 5672 -> [["rabbit1",5672], ["rabbit2",5672]]
|
|
121
|
+
# host: "rabbit1", port: 5672 (default) -> [["rabbit1",5672]]
|
|
122
|
+
def self.address_list(host:, port:, hosts: nil, addresses: nil)
|
|
123
|
+
if addresses && !addresses.empty?
|
|
124
|
+
addresses.map do |addr|
|
|
125
|
+
h, p = addr.to_s.split(":", 2)
|
|
126
|
+
[h, p ? p.to_i : port]
|
|
127
|
+
end
|
|
128
|
+
elsif hosts && !hosts.empty?
|
|
129
|
+
hosts.map { |h| [h.to_s, port] }
|
|
130
|
+
else
|
|
131
|
+
[[host.to_s, port]]
|
|
132
|
+
end
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
def initialize(
|
|
136
|
+
host: "localhost",
|
|
137
|
+
port: 5672,
|
|
138
|
+
hosts: nil,
|
|
139
|
+
addresses: nil,
|
|
140
|
+
hosts_shuffle_strategy: :shuffle,
|
|
141
|
+
vhost: "/",
|
|
142
|
+
username: "guest",
|
|
143
|
+
password: "guest",
|
|
144
|
+
tls: false,
|
|
145
|
+
tls_context: nil,
|
|
146
|
+
tls_cert: nil,
|
|
147
|
+
tls_key: nil,
|
|
148
|
+
tls_ca_certificates: nil,
|
|
149
|
+
verify_peer: true,
|
|
150
|
+
tls_min_version: :TLS1_2,
|
|
151
|
+
heartbeat: 60,
|
|
152
|
+
tcp_user_timeout: nil,
|
|
153
|
+
frame_max: 131_072,
|
|
154
|
+
channel_max: 2047,
|
|
155
|
+
connect_timeout: CONNECT_TIMEOUT,
|
|
156
|
+
rpc_timeout: RPC_TIMEOUT,
|
|
157
|
+
auth_mechanism: nil,
|
|
158
|
+
connection_name: nil,
|
|
159
|
+
auto_recover: true,
|
|
160
|
+
recovery_attempts: nil,
|
|
161
|
+
recovery_interval: RECOVERY_INITIAL,
|
|
162
|
+
recovery_max_interval: RECOVERY_MAX,
|
|
163
|
+
recover_topology: true,
|
|
164
|
+
topology_recovery_filter: nil,
|
|
165
|
+
instrumenter: nil,
|
|
166
|
+
notifier: nil,
|
|
167
|
+
logger: Log.new
|
|
168
|
+
)
|
|
169
|
+
@addresses = self.class.address_list(host: host, port: port, hosts: hosts, addresses: addresses)
|
|
170
|
+
@host = @addresses.first[0]
|
|
171
|
+
@port = @addresses.first[1]
|
|
172
|
+
@hosts_shuffle_strategy = hosts_shuffle_strategy
|
|
173
|
+
@vhost = vhost
|
|
174
|
+
@username = username
|
|
175
|
+
@password = password
|
|
176
|
+
# Certificate material or a ready context means TLS, whatever tls: says.
|
|
177
|
+
@tls = tls || !(tls_context || tls_cert || tls_key || tls_ca_certificates).nil?
|
|
178
|
+
@tls_context = tls_context
|
|
179
|
+
@tls_cert = tls_cert
|
|
180
|
+
@tls_key = tls_key
|
|
181
|
+
@tls_ca_certificates = tls_ca_certificates
|
|
182
|
+
@verify_peer = verify_peer
|
|
183
|
+
@tls_min_version = tls_min_version
|
|
184
|
+
@heartbeat = heartbeat
|
|
185
|
+
@tcp_user_timeout = tcp_user_timeout
|
|
186
|
+
@frame_max = frame_max
|
|
187
|
+
@channel_max = channel_max
|
|
188
|
+
@connect_timeout = connect_timeout
|
|
189
|
+
@rpc_timeout = rpc_timeout
|
|
190
|
+
@auth_mechanism = auth_mechanism
|
|
191
|
+
@connection_name = connection_name
|
|
192
|
+
@auto_recover = auto_recover
|
|
193
|
+
@recovery_attempts = recovery_attempts # nil = unlimited
|
|
194
|
+
@recovery_interval = recovery_interval
|
|
195
|
+
@recovery_max_interval = recovery_max_interval
|
|
196
|
+
@recover_topology = recover_topology
|
|
197
|
+
@topology = TopologyRegistry.new(filter: topology_recovery_filter)
|
|
198
|
+
@logger = logger
|
|
199
|
+
@notifier = notifier || Notifier.new(logger: logger)
|
|
200
|
+
@notifier.subscribe { |name, payload| instrumenter.call(name, payload) } if instrumenter
|
|
201
|
+
|
|
202
|
+
@state = :closed
|
|
203
|
+
@frame_io = nil
|
|
204
|
+
@channels = {} # channel_id => Channel; a closed channel's id is free again
|
|
205
|
+
@channel_mutex = Async::Semaphore.new(1)
|
|
206
|
+
@negotiated_hb = nil
|
|
207
|
+
@negotiated_fm = nil
|
|
208
|
+
@negotiated_cmax = nil
|
|
209
|
+
@recovery_in_progress = false
|
|
210
|
+
@recovery_interrupted = false
|
|
211
|
+
@closed_by_user = false
|
|
212
|
+
@connecting = false
|
|
213
|
+
@open_condition = nil
|
|
214
|
+
@last_frame_at = nil
|
|
215
|
+
@heartbeat_task = nil
|
|
216
|
+
@channel0_task = nil # drains channel-0 queue; handles connection.blocked/unblocked
|
|
217
|
+
@recovery_task = nil
|
|
218
|
+
@recovery_wakeup = nil # Async::Condition to interrupt the retry sleep
|
|
219
|
+
@session_root_task = nil # top-level task; recovery is spawned here so it
|
|
220
|
+
# survives reader-task Cancel propagation
|
|
221
|
+
@on_blocked = nil
|
|
222
|
+
@on_unblocked = nil
|
|
223
|
+
@update_secret_condition = nil
|
|
224
|
+
@on_recovery_attempt = nil
|
|
225
|
+
@on_recovery = nil
|
|
226
|
+
@on_recovery_exhausted = nil
|
|
227
|
+
@on_connection_lost = nil
|
|
228
|
+
end
|
|
229
|
+
|
|
230
|
+
# Connect and complete AMQP handshake. Raises ConnectionTimeoutError if
|
|
231
|
+
# the handshake does not complete within +connect_timeout+ seconds.
|
|
232
|
+
# Raises NotOpenError if already connected.
|
|
233
|
+
def connect
|
|
234
|
+
raise NotOpenError, "Session is already connected" if open?
|
|
235
|
+
@connecting = true
|
|
236
|
+
@closed_by_user = false # a previous failed connect must not disable recovery
|
|
237
|
+
@session_root_task = Async::Task.current
|
|
238
|
+
|
|
239
|
+
last_error = nil
|
|
240
|
+
started_at = instrument_clock
|
|
241
|
+
shuffled_addresses.each do |target_host, target_port|
|
|
242
|
+
begin
|
|
243
|
+
Async::Task.current.with_timeout(@connect_timeout) do
|
|
244
|
+
raw_socket = open_socket(target_host, target_port)
|
|
245
|
+
@frame_io = build_frame_io(raw_socket)
|
|
246
|
+
@frame_io.start(spawn: method(:spawn_background))
|
|
247
|
+
handshake
|
|
248
|
+
start_heartbeat_task
|
|
249
|
+
start_channel0_monitor_task
|
|
250
|
+
@host = target_host
|
|
251
|
+
@port = target_port
|
|
252
|
+
@state = :open
|
|
253
|
+
@channel_ids = fresh_channel_ids
|
|
254
|
+
end
|
|
255
|
+
@connecting = false
|
|
256
|
+
instrument("connection.open") do
|
|
257
|
+
{ host: @host, port: @port, vhost: @vhost, tls: @tls, heartbeat: @negotiated_hb,
|
|
258
|
+
frame_max: @negotiated_fm, channel_max: @negotiated_cmax,
|
|
259
|
+
duration: started_at ? instrument_elapsed(started_at) : nil }
|
|
260
|
+
end
|
|
261
|
+
return
|
|
262
|
+
rescue Async::TimeoutError => e
|
|
263
|
+
@frame_io&.stop rescue nil
|
|
264
|
+
@frame_io = nil
|
|
265
|
+
last_error = e
|
|
266
|
+
rescue SystemCallError, EOFError, IOError, SocketError => e
|
|
267
|
+
# TCP connect failed (refused, unreachable, timed out) or the peer
|
|
268
|
+
# dropped the connection mid-handshake (reset, EOF, shutdown).
|
|
269
|
+
@frame_io&.stop rescue nil
|
|
270
|
+
@frame_io = nil
|
|
271
|
+
last_error = e
|
|
272
|
+
rescue OpenSSL::SSL::SSLError => e
|
|
273
|
+
@frame_io&.stop rescue nil
|
|
274
|
+
@frame_io = nil
|
|
275
|
+
last_error = e
|
|
276
|
+
rescue AuthenticationError
|
|
277
|
+
# Credentials or vhost access refused: other nodes will say the same.
|
|
278
|
+
cleanup_after_failed_connect
|
|
279
|
+
raise
|
|
280
|
+
rescue ConnectionError, ChannelError => e
|
|
281
|
+
# The broker refused the handshake (e.g. 530 NOT_ALLOWED for a missing
|
|
282
|
+
# vhost, 320 CONNECTION_FORCED from a node in maintenance). Stop the
|
|
283
|
+
# reader/writer tasks on this socket and try the next address.
|
|
284
|
+
@frame_io&.stop rescue nil
|
|
285
|
+
@frame_io = nil
|
|
286
|
+
last_error = e
|
|
287
|
+
end
|
|
288
|
+
end
|
|
289
|
+
|
|
290
|
+
# All addresses exhausted
|
|
291
|
+
cleanup_after_failed_connect
|
|
292
|
+
tried = @addresses.map { |h, p| "#{h}:#{p}" }.join(", ")
|
|
293
|
+
case last_error
|
|
294
|
+
when Async::TimeoutError
|
|
295
|
+
raise ConnectionTimeoutError, "AMQP handshake did not complete within #{@connect_timeout}s (tried #{tried})"
|
|
296
|
+
when OpenSSL::SSL::SSLError
|
|
297
|
+
raise ConnectionTimeoutError, "TLS handshake failed (tried #{tried}) — #{last_error.message}"
|
|
298
|
+
when ConnectionError, ChannelError
|
|
299
|
+
raise last_error
|
|
300
|
+
else
|
|
301
|
+
raise ConnectionTimeoutError, "Could not connect to any host (tried #{tried}) — #{last_error&.message}"
|
|
302
|
+
end
|
|
303
|
+
end
|
|
304
|
+
|
|
305
|
+
def open?
|
|
306
|
+
@state == :open
|
|
307
|
+
end
|
|
308
|
+
|
|
309
|
+
def closed?
|
|
310
|
+
@state == :closed
|
|
311
|
+
end
|
|
312
|
+
|
|
313
|
+
# Internal. Run a long-lived background loop (channel dispatch, heartbeat,
|
|
314
|
+
# channel-0 monitor).
|
|
315
|
+
#
|
|
316
|
+
# These must not become children of whichever task happened to open the
|
|
317
|
+
# channel or the session: a short-lived caller that finishes, or is stopped,
|
|
318
|
+
# would silently take its dispatch loop with it while the channel still
|
|
319
|
+
# reported itself open. Parent them at the reactor instead, and mark them
|
|
320
|
+
# transient so they never hold the reactor open by themselves — #close
|
|
321
|
+
# stops them explicitly.
|
|
322
|
+
#
|
|
323
|
+
# @api private
|
|
324
|
+
def spawn_background(&block)
|
|
325
|
+
# Looked up per call, never memoised: a Session reused across two
|
|
326
|
+
# separate Sync/Async reactors would otherwise keep spawning onto the
|
|
327
|
+
# first, finished one.
|
|
328
|
+
Async::Task.current.root.async(transient: true, &block)
|
|
329
|
+
end
|
|
330
|
+
|
|
331
|
+
# Close the session gracefully. Also works if recovery is in progress.
|
|
332
|
+
def close
|
|
333
|
+
return if closed?
|
|
334
|
+
@closed_by_user = true
|
|
335
|
+
@recovery_wakeup&.signal rescue nil
|
|
336
|
+
# Unblock any fibers waiting on channel replies (e.g. wait_for inside
|
|
337
|
+
# reopen_after_recovery) so that @recovery_task.cancel below can actually
|
|
338
|
+
# terminate the task rather than leaving it stuck on an Async::Condition.
|
|
339
|
+
conn_error = ConnectionError.new(code: 0, text: "Session closed by user")
|
|
340
|
+
@channels.values.each { |ch| ch.mark_closed!(conn_error) rescue nil }
|
|
341
|
+
if open?
|
|
342
|
+
@state = :closing
|
|
343
|
+
@channel0_task&.cancel rescue nil
|
|
344
|
+
send_connection_close rescue nil
|
|
345
|
+
end
|
|
346
|
+
@heartbeat_task&.cancel rescue nil
|
|
347
|
+
@channel0_task&.cancel rescue nil
|
|
348
|
+
@recovery_task&.cancel rescue nil
|
|
349
|
+
@frame_io&.stop rescue nil
|
|
350
|
+
@state = :closed
|
|
351
|
+
instrument("connection.closed") { { reason: :user, host: @host, port: @port } }
|
|
352
|
+
end
|
|
353
|
+
|
|
354
|
+
# Open a new channel. Returns an AsyncRabbitMQ::Channel.
|
|
355
|
+
# +pool_size+ bounds concurrent consumer-handler fibers on the channel
|
|
356
|
+
# (Bunny-parity: default 1). basic_qos will auto-adjust it to prefetch_count.
|
|
357
|
+
def open_channel(pool_size: 1)
|
|
358
|
+
raise NotOpenError, "Session is not open" unless open?
|
|
359
|
+
channel = @channel_mutex.acquire do
|
|
360
|
+
channel_id = next_channel_id
|
|
361
|
+
Channel.new(channel_id, self, @frame_io, frame_max: @negotiated_fm || @frame_max, logger: @logger,
|
|
362
|
+
pool_size: pool_size, rpc_timeout: @rpc_timeout).tap { |ch| @channels[channel_id] = ch }
|
|
363
|
+
end
|
|
364
|
+
channel.open
|
|
365
|
+
channel
|
|
366
|
+
end
|
|
367
|
+
|
|
368
|
+
# Open a channel, yield it to the block, and ensure it is closed afterward.
|
|
369
|
+
def with_channel(pool_size: 1)
|
|
370
|
+
ch = open_channel(pool_size: pool_size)
|
|
371
|
+
begin
|
|
372
|
+
yield ch
|
|
373
|
+
ensure
|
|
374
|
+
ch.close rescue nil
|
|
375
|
+
end
|
|
376
|
+
end
|
|
377
|
+
|
|
378
|
+
# Rotate the secret this connection authenticated with, without
|
|
379
|
+
# reconnecting (connection.update-secret, RabbitMQ 3.8+). Used with the
|
|
380
|
+
# OAuth 2 auth backend to hand the broker a refreshed access token before
|
|
381
|
+
# the current one expires. The new secret is also used for reconnects.
|
|
382
|
+
# Raises ConnectionError if the broker refuses (e.g. an auth backend that
|
|
383
|
+
# does not support secret updates closes the connection).
|
|
384
|
+
def update_secret(new_secret, reason = "secret update")
|
|
385
|
+
raise NotOpenError, "Session is not open" unless open?
|
|
386
|
+
|
|
387
|
+
cond = @update_secret_condition = Async::Condition.new
|
|
388
|
+
@frame_io.write_frame(AMQ::Protocol::Connection::UpdateSecret.encode(new_secret, reason).encode)
|
|
389
|
+
result = Async::Task.current.with_timeout(@rpc_timeout || RPC_TIMEOUT) { cond.wait }
|
|
390
|
+
raise result if result.is_a?(Exception)
|
|
391
|
+
|
|
392
|
+
@password = new_secret
|
|
393
|
+
true
|
|
394
|
+
rescue Async::TimeoutError
|
|
395
|
+
raise RpcTimeoutError, "No reply to connection.update-secret within #{@rpc_timeout || RPC_TIMEOUT}s"
|
|
396
|
+
ensure
|
|
397
|
+
@update_secret_condition = nil
|
|
398
|
+
end
|
|
399
|
+
|
|
400
|
+
# Check whether a queue exists on the broker without creating it.
|
|
401
|
+
# Opens a temporary channel and performs a passive declare.
|
|
402
|
+
def queue_exists?(name)
|
|
403
|
+
with_channel { |ch| ch.queue(name, passive: true) }
|
|
404
|
+
true
|
|
405
|
+
rescue ChannelError
|
|
406
|
+
false
|
|
407
|
+
end
|
|
408
|
+
|
|
409
|
+
# Check whether an exchange exists on the broker without creating it.
|
|
410
|
+
# Opens a temporary channel and performs a passive declare.
|
|
411
|
+
def exchange_exists?(name)
|
|
412
|
+
with_channel { |ch| ch.exchange(name, passive: true) }
|
|
413
|
+
true
|
|
414
|
+
rescue ChannelError
|
|
415
|
+
false
|
|
416
|
+
end
|
|
417
|
+
|
|
418
|
+
# Register a callback invoked when the broker sends connection.blocked.
|
|
419
|
+
# The block receives the reason string from the broker.
|
|
420
|
+
def on_blocked(&block)
|
|
421
|
+
@on_blocked = block
|
|
422
|
+
end
|
|
423
|
+
|
|
424
|
+
# Register a callback invoked when the broker sends connection.unblocked.
|
|
425
|
+
def on_unblocked(&block)
|
|
426
|
+
@on_unblocked = block
|
|
427
|
+
end
|
|
428
|
+
|
|
429
|
+
# Register a callback invoked at the start of each recovery attempt.
|
|
430
|
+
# The block receives the attempt number (1-based).
|
|
431
|
+
def on_recovery_attempt(&block)
|
|
432
|
+
@on_recovery_attempt = block
|
|
433
|
+
end
|
|
434
|
+
|
|
435
|
+
# Register a callback invoked after recovery succeeds.
|
|
436
|
+
# The block receives the session.
|
|
437
|
+
def on_recovery(&block)
|
|
438
|
+
@on_recovery = block
|
|
439
|
+
end
|
|
440
|
+
|
|
441
|
+
# Subscribe to structured events from this session and its channels. The
|
|
442
|
+
# block receives (name, payload). +pattern+ is nil for every event, a
|
|
443
|
+
# String for one name or a prefix ending in a dot, or a Regexp. Returns a
|
|
444
|
+
# handle for notifier.unsubscribe. See Notifier::EVENTS for the taxonomy.
|
|
445
|
+
#
|
|
446
|
+
# session.on_event("message.") { |name, payload| statsd.increment(name) }
|
|
447
|
+
def on_event(pattern = nil, &block)
|
|
448
|
+
@notifier.subscribe(pattern, &block)
|
|
449
|
+
end
|
|
450
|
+
|
|
451
|
+
# Register a callback invoked when recovery attempts are exhausted.
|
|
452
|
+
# Only fires when recovery_attempts is set to a finite number.
|
|
453
|
+
# The block receives the session.
|
|
454
|
+
def on_recovery_exhausted(&block)
|
|
455
|
+
@on_recovery_exhausted = block
|
|
456
|
+
end
|
|
457
|
+
|
|
458
|
+
# Register a callback invoked when the connection is lost, before
|
|
459
|
+
# recovery starts (with auto_recover: false, after the channels have
|
|
460
|
+
# been closed). The block receives the session and the error.
|
|
461
|
+
def on_connection_lost(&block)
|
|
462
|
+
@on_connection_lost = block
|
|
463
|
+
end
|
|
464
|
+
|
|
465
|
+
# The channels open or recovering on this connection.
|
|
466
|
+
def channels
|
|
467
|
+
@channels.values
|
|
468
|
+
end
|
|
469
|
+
|
|
470
|
+
def channel_count
|
|
471
|
+
@channels.size
|
|
472
|
+
end
|
|
473
|
+
|
|
474
|
+
# Record a rotated secret for the next connect or reconnect without
|
|
475
|
+
# sending it: #update_secret does this itself on a live connection.
|
|
476
|
+
# For a session that is down while the secret rotates.
|
|
477
|
+
def store_secret(new_secret)
|
|
478
|
+
@password = new_secret
|
|
479
|
+
end
|
|
480
|
+
|
|
481
|
+
# Called by Channel when it closes itself.
|
|
482
|
+
def channel_closed(channel_id)
|
|
483
|
+
@channel_mutex.acquire do
|
|
484
|
+
@channels.delete(channel_id)
|
|
485
|
+
@channel_ids&.release(channel_id)
|
|
486
|
+
end
|
|
487
|
+
@frame_io&.unregister_channel(channel_id)
|
|
488
|
+
end
|
|
489
|
+
|
|
490
|
+
# Put a channel the broker closed back into the channel table and reopen
|
|
491
|
+
# it on the current connection under its original id (Channel#reopen).
|
|
492
|
+
def reopen_channel(channel, state: :open)
|
|
493
|
+
raise NotOpenError, "Session is not open" unless open?
|
|
494
|
+
@channel_mutex.acquire do
|
|
495
|
+
existing = @channels[channel.channel_id]
|
|
496
|
+
if existing && !existing.equal?(channel)
|
|
497
|
+
raise Error, "Channel id #{channel.channel_id} is in use by another channel"
|
|
498
|
+
end
|
|
499
|
+
@channel_ids ||= fresh_channel_ids
|
|
500
|
+
@channel_ids.reserve(channel.channel_id) unless existing
|
|
501
|
+
@channels[channel.channel_id] = channel
|
|
502
|
+
end
|
|
503
|
+
channel.reopen_on(@frame_io, state: state)
|
|
504
|
+
end
|
|
505
|
+
|
|
506
|
+
# Called by a channel when topology recovery re-declared a server-named
|
|
507
|
+
# queue under a new name: update the registry and every consumer of it.
|
|
508
|
+
def queue_renamed(old_name, new_name)
|
|
509
|
+
@logger.info("Server-named queue #{old_name} recovered as #{new_name}")
|
|
510
|
+
@topology.rename_queue(old_name, new_name)
|
|
511
|
+
@channels.values.each { |ch| ch.rename_consumer_queue(old_name, new_name) }
|
|
512
|
+
end
|
|
513
|
+
|
|
514
|
+
# --- Async::Pool resource interface ---
|
|
515
|
+
|
|
516
|
+
# Can this session be returned to the pool?
|
|
517
|
+
def reusable?
|
|
518
|
+
open? && !@recovery_in_progress
|
|
519
|
+
end
|
|
520
|
+
|
|
521
|
+
# Is the underlying socket alive?
|
|
522
|
+
def viable?
|
|
523
|
+
open?
|
|
524
|
+
end
|
|
525
|
+
|
|
526
|
+
# Max concurrent channels (negotiated with broker, or AMQP max 2047).
|
|
527
|
+
def concurrency
|
|
528
|
+
@negotiated_cmax || 2047
|
|
529
|
+
end
|
|
530
|
+
|
|
531
|
+
# --- Internal ---
|
|
532
|
+
|
|
533
|
+
# Called by FrameIO when a connection-level error triggers recovery.
|
|
534
|
+
def trigger_recovery(error, from: nil)
|
|
535
|
+
return if @closed_by_user
|
|
536
|
+
# A frame_io that a later connection attempt has already replaced is
|
|
537
|
+
# still draining its dead socket, and its errors are not the live
|
|
538
|
+
# connection's. Letting them through made a reconnect look like it had
|
|
539
|
+
# dropped again the moment it succeeded.
|
|
540
|
+
return if from && !@frame_io.equal?(from)
|
|
541
|
+
|
|
542
|
+
if @connecting
|
|
543
|
+
# The handshake is in flight in another fiber; hand it the IO error so
|
|
544
|
+
# it fails now (as a connect failure) rather than sitting in
|
|
545
|
+
# wait_channel0_method until the connect timeout expires.
|
|
546
|
+
q0 = @frame_io&.channel_queue(0)
|
|
547
|
+
q0&.push([:method, error]) rescue nil
|
|
548
|
+
return
|
|
549
|
+
end
|
|
550
|
+
|
|
551
|
+
instrument("connection.lost") do
|
|
552
|
+
{ host: @host, port: @port, error: error.class.name, message: error.message,
|
|
553
|
+
recovering: @auto_recover }
|
|
554
|
+
end
|
|
555
|
+
|
|
556
|
+
unless @auto_recover
|
|
557
|
+
@state = :closed
|
|
558
|
+
conn_error = ConnectionError.new(code: 0, text: "Connection lost: #{error.message}")
|
|
559
|
+
@channels.values.each { |ch| ch.mark_closed!(conn_error) rescue nil }
|
|
560
|
+
connection_lost(error)
|
|
561
|
+
@frame_io&.stop rescue nil
|
|
562
|
+
return
|
|
563
|
+
end
|
|
564
|
+
|
|
565
|
+
@logger.warn("Connection lost (#{error.class}: #{error.message}). Starting recovery...")
|
|
566
|
+
|
|
567
|
+
if @recovery_in_progress
|
|
568
|
+
# Connection died again while a recovery attempt is in progress.
|
|
569
|
+
# Unblock any fiber stuck in wait_channel0_method or channel wait_for
|
|
570
|
+
# so that the current recover_loop iteration fails fast and retries.
|
|
571
|
+
recovery_error = ConnectionError.new(code: 0, text: "Connection lost during recovery")
|
|
572
|
+
@recovery_interrupted = true
|
|
573
|
+
q0 = @frame_io&.channel_queue(0)
|
|
574
|
+
q0&.push([:method, recovery_error]) rescue nil
|
|
575
|
+
@channels.each_value { |ch| ch.interrupt_wait!(recovery_error) rescue nil }
|
|
576
|
+
return
|
|
577
|
+
end
|
|
578
|
+
|
|
579
|
+
@recovery_in_progress = true
|
|
580
|
+
@state = :recovering
|
|
581
|
+
|
|
582
|
+
# Stop the channel-0 monitor so it doesn't race on the stale queue.
|
|
583
|
+
# Also unblock any fibers stuck in write_frame waiting on connection.blocked.
|
|
584
|
+
@channel0_task&.cancel rescue nil
|
|
585
|
+
@channel0_task = nil
|
|
586
|
+
@frame_io&.set_unblocked rescue nil
|
|
587
|
+
|
|
588
|
+
# Waits in flight raise ConnectionError instead of hanging; new operations
|
|
589
|
+
# on the channels park until they are reopened after reconnect.
|
|
590
|
+
conn_error = ConnectionError.new(code: 0, text: "Connection lost: #{error.message}")
|
|
591
|
+
@channels.values.each { |ch| ch.mark_recovering!(conn_error) rescue nil }
|
|
592
|
+
@update_secret_condition&.signal(conn_error)
|
|
593
|
+
# Tell the owner now, while the channels are still in the table: a
|
|
594
|
+
# Cluster in :drop mode gives them up here, before they are reopened.
|
|
595
|
+
connection_lost(error)
|
|
596
|
+
|
|
597
|
+
# Schedule recover_loop BEFORE stopping old frame_io. old_io.stop
|
|
598
|
+
# cancels reader/writer tasks; if we ARE the reader task, cancel raises
|
|
599
|
+
# Async::Cancel (< Exception) which bypasses `rescue nil` and propagates,
|
|
600
|
+
# so anything after stop might not run.
|
|
601
|
+
# Parent recovery at the reactor, not at Async::Task.current — which here
|
|
602
|
+
# is usually the reader task, whose Cancel would propagate into a child
|
|
603
|
+
# recovery task — and not at the task that called connect either, since
|
|
604
|
+
# that one may be short-lived while the session outlives it.
|
|
605
|
+
@recovery_task = spawn_background { recover_loop }
|
|
606
|
+
|
|
607
|
+
# Stop the old frame_io: closes the dead socket, pushes nil to channel
|
|
608
|
+
# queues (unblocking wait_channel0_method), and cancels writer task.
|
|
609
|
+
# Reader task cancel may raise Async::Cancel — recovery is already scheduled.
|
|
610
|
+
old_io = @frame_io
|
|
611
|
+
@frame_io = nil
|
|
612
|
+
old_io&.stop rescue nil
|
|
613
|
+
end
|
|
614
|
+
|
|
615
|
+
def frame_max
|
|
616
|
+
@negotiated_fm || @frame_max
|
|
617
|
+
end
|
|
618
|
+
|
|
619
|
+
private
|
|
620
|
+
|
|
621
|
+
# Hand the connection-lost callback its error without letting a failure
|
|
622
|
+
# in it stop recovery.
|
|
623
|
+
def connection_lost(error)
|
|
624
|
+
@on_connection_lost&.call(self, error)
|
|
625
|
+
rescue => e
|
|
626
|
+
@logger.error("on_connection_lost callback failed: #{e.class}: #{e.message}")
|
|
627
|
+
end
|
|
628
|
+
|
|
629
|
+
# Return the address list in the order they should be tried for this
|
|
630
|
+
# connect/recovery cycle.
|
|
631
|
+
def shuffled_addresses
|
|
632
|
+
case @hosts_shuffle_strategy
|
|
633
|
+
when :shuffle then @addresses.shuffle
|
|
634
|
+
when :none then @addresses.dup
|
|
635
|
+
when Proc then @hosts_shuffle_strategy.call(@addresses)
|
|
636
|
+
else @addresses.shuffle
|
|
637
|
+
end
|
|
638
|
+
end
|
|
639
|
+
|
|
640
|
+
# Stop any in-flight frame_io / recovery tasks spawned during a failed connect.
|
|
641
|
+
def cleanup_after_failed_connect
|
|
642
|
+
@closed_by_user = true
|
|
643
|
+
@connecting = false
|
|
644
|
+
@recovery_task&.cancel rescue nil
|
|
645
|
+
@recovery_task = nil
|
|
646
|
+
@heartbeat_task&.cancel rescue nil
|
|
647
|
+
@channel0_task&.cancel rescue nil
|
|
648
|
+
@frame_io&.stop rescue nil
|
|
649
|
+
@frame_io = nil
|
|
650
|
+
@state = :closed
|
|
651
|
+
end
|
|
652
|
+
|
|
653
|
+
def open_socket(target_host = @host, target_port = @port)
|
|
654
|
+
if @tls
|
|
655
|
+
require "openssl"
|
|
656
|
+
ctx = @tls_context || build_tls_context
|
|
657
|
+
raw = TCPSocket.new(target_host, target_port)
|
|
658
|
+
# SSLSocket does not close the socket it wraps when the handshake fails,
|
|
659
|
+
# and a failed connect has no frame_io to stop, so without this a broker
|
|
660
|
+
# with a bad certificate leaks one descriptor per recovery attempt.
|
|
661
|
+
handshaked = false
|
|
662
|
+
begin
|
|
663
|
+
ssl = OpenSSL::SSL::SSLSocket.new(raw, ctx)
|
|
664
|
+
# Without this, closing the SSL socket sends close_notify but leaves
|
|
665
|
+
# the TCP socket open until the GC finalises it, so every amqps
|
|
666
|
+
# reconnect strands a connection (Amazon MQ, for example).
|
|
667
|
+
ssl.sync_close = true
|
|
668
|
+
ssl.hostname = target_host
|
|
669
|
+
ssl.connect
|
|
670
|
+
handshaked = true
|
|
671
|
+
apply_socket_timeouts(raw)
|
|
672
|
+
ssl
|
|
673
|
+
ensure
|
|
674
|
+
raw.close unless handshaked
|
|
675
|
+
end
|
|
676
|
+
else
|
|
677
|
+
TCPSocket.new(target_host, target_port).tap { |sock| apply_socket_timeouts(sock) }
|
|
678
|
+
end
|
|
679
|
+
end
|
|
680
|
+
|
|
681
|
+
# Bound how long the kernel will retransmit unacknowledged data before it
|
|
682
|
+
# gives up on the socket. Without it a peer that disappears mid-write (a
|
|
683
|
+
# network partition with a full send buffer) is only noticed when the TCP
|
|
684
|
+
# retransmit timer expires, which on Linux is around 15 minutes — far
|
|
685
|
+
# longer than the heartbeat timeout we promise callers.
|
|
686
|
+
#
|
|
687
|
+
# TCP_USER_TIMEOUT is Linux-only; elsewhere the heartbeat task's own write
|
|
688
|
+
# timeout is the backstop, so a missing constant is not an error.
|
|
689
|
+
def apply_socket_timeouts(sock)
|
|
690
|
+
millis = @tcp_user_timeout
|
|
691
|
+
millis ||= (@heartbeat.to_i > 0 ? @heartbeat.to_i * 2 * 1000 : nil)
|
|
692
|
+
return unless millis && millis > 0
|
|
693
|
+
return unless Socket.const_defined?(:TCP_USER_TIMEOUT)
|
|
694
|
+
|
|
695
|
+
sock.setsockopt(Socket::IPPROTO_TCP, Socket::TCP_USER_TIMEOUT, millis)
|
|
696
|
+
rescue StandardError => e
|
|
697
|
+
# Unsupported on this platform or socket type; the heartbeat still covers us.
|
|
698
|
+
@logger.debug("Could not set TCP_USER_TIMEOUT: #{e.class}: #{e.message}")
|
|
699
|
+
end
|
|
700
|
+
|
|
701
|
+
def build_tls_context
|
|
702
|
+
TLS.context(cert: @tls_cert, key: @tls_key, ca_certificates: @tls_ca_certificates,
|
|
703
|
+
verify_peer: @verify_peer, min_version: @tls_min_version)
|
|
704
|
+
end
|
|
705
|
+
|
|
706
|
+
def build_frame_io(raw_socket)
|
|
707
|
+
io = FrameIO.new(raw_socket, logger: @logger)
|
|
708
|
+
# Register channel 0 for connection-level frames
|
|
709
|
+
io.register_channel(0)
|
|
710
|
+
|
|
711
|
+
# Override trigger_recovery to delegate to Session
|
|
712
|
+
session = self
|
|
713
|
+
io.define_singleton_method(:trigger_recovery) { |error| session.trigger_recovery(error, from: io) }
|
|
714
|
+
|
|
715
|
+
# Update heartbeat timestamp on every received frame
|
|
716
|
+
io.on_frame = -> { @last_frame_at = Process.clock_gettime(Process::CLOCK_MONOTONIC) }
|
|
717
|
+
io
|
|
718
|
+
end
|
|
719
|
+
|
|
720
|
+
def handshake
|
|
721
|
+
# Send AMQP protocol header directly on the socket
|
|
722
|
+
@frame_io.instance_variable_get(:@socket).write(PROTOCOL_HEADER)
|
|
723
|
+
|
|
724
|
+
# connection.start — negotiate SASL mechanism
|
|
725
|
+
start = wait_channel0_method(AMQ::Protocol::Connection::Start)
|
|
726
|
+
sasl = SASL.negotiate(
|
|
727
|
+
start.mechanisms,
|
|
728
|
+
preferred: @auth_mechanism,
|
|
729
|
+
username: @username,
|
|
730
|
+
password: @password
|
|
731
|
+
)
|
|
732
|
+
send_connection_start_ok(sasl)
|
|
733
|
+
|
|
734
|
+
# connection.tune (broker may send connection.secure challenges first)
|
|
735
|
+
msg = wait_channel0_method(AMQ::Protocol::Connection::Tune, AMQ::Protocol::Connection::Secure)
|
|
736
|
+
while msg.is_a?(AMQ::Protocol::Connection::Secure)
|
|
737
|
+
@logger.debug("Received connection.secure SASL challenge for #{sasl.mechanism_name}")
|
|
738
|
+
@frame_io.write_frame(AMQ::Protocol::Connection::SecureOk.encode(sasl.challenge_response(msg.challenge)).encode)
|
|
739
|
+
msg = wait_channel0_method(AMQ::Protocol::Connection::Tune, AMQ::Protocol::Connection::Secure)
|
|
740
|
+
end
|
|
741
|
+
@negotiated_hb = negotiate_heartbeat(msg.heartbeat)
|
|
742
|
+
@negotiated_fm = negotiate_frame_max(msg.frame_max)
|
|
743
|
+
@negotiated_cmax = negotiate_channel_max(msg.channel_max)
|
|
744
|
+
send_connection_tune_ok
|
|
745
|
+
|
|
746
|
+
# connection.open
|
|
747
|
+
send_connection_open
|
|
748
|
+
wait_channel0_method(AMQ::Protocol::Connection::OpenOk)
|
|
749
|
+
end
|
|
750
|
+
|
|
751
|
+
def wait_channel0_method(*expected_classes)
|
|
752
|
+
queue = @frame_io.channel_queue(0)
|
|
753
|
+
loop do
|
|
754
|
+
msg = queue.pop
|
|
755
|
+
# nil sentinel — pushed by frame_io.stop or interrupt during recovery
|
|
756
|
+
raise ConnectionError, "Connection closed while waiting for #{expected_classes.join(', ')}" if msg.nil?
|
|
757
|
+
next unless msg[0] == :method
|
|
758
|
+
method = msg[1]
|
|
759
|
+
# An exception pushed directly by trigger_recovery to unblock this wait
|
|
760
|
+
# (ConnectionError during recovery, or the raw IO error during connect).
|
|
761
|
+
raise method if method.is_a?(Exception)
|
|
762
|
+
if expected_classes.any? { |c| method.is_a?(c) }
|
|
763
|
+
return method
|
|
764
|
+
elsif method.is_a?(AMQ::Protocol::Connection::Close)
|
|
765
|
+
code = method.reply_code
|
|
766
|
+
text = method.reply_text
|
|
767
|
+
# connection.close is always connection-level. 403 ACCESS_REFUSED here
|
|
768
|
+
# means the credentials or the vhost access were rejected.
|
|
769
|
+
if code == 403
|
|
770
|
+
raise AuthenticationError.new(code: code, text: text)
|
|
771
|
+
else
|
|
772
|
+
raise ConnectionError.new(code: code, text: text)
|
|
773
|
+
end
|
|
774
|
+
end
|
|
775
|
+
end
|
|
776
|
+
end
|
|
777
|
+
|
|
778
|
+
def send_connection_start_ok(sasl)
|
|
779
|
+
props = {
|
|
780
|
+
"product" => "async-rabbitmq",
|
|
781
|
+
"version" => AsyncRabbitMQ::VERSION,
|
|
782
|
+
"platform" => "Ruby #{RUBY_VERSION}",
|
|
783
|
+
"information" => "https://github.com/womblep/async-rabbitmq",
|
|
784
|
+
# Extensions this client understands. authentication_failure_close makes
|
|
785
|
+
# RabbitMQ answer bad credentials with connection.close 403 instead of
|
|
786
|
+
# silently dropping the TCP connection (https://www.rabbitmq.com/docs/auth-notification).
|
|
787
|
+
"capabilities" => {
|
|
788
|
+
"publisher_confirms" => true,
|
|
789
|
+
"consumer_cancel_notify" => true,
|
|
790
|
+
"exchange_exchange_bindings" => true,
|
|
791
|
+
"basic.nack" => true,
|
|
792
|
+
"connection.blocked" => true,
|
|
793
|
+
"authentication_failure_close" => true,
|
|
794
|
+
},
|
|
795
|
+
}
|
|
796
|
+
props["connection_name"] = @connection_name if @connection_name
|
|
797
|
+
@frame_io.write_frame(
|
|
798
|
+
AMQ::Protocol::Connection::StartOk.encode(
|
|
799
|
+
props,
|
|
800
|
+
sasl.mechanism_name,
|
|
801
|
+
sasl.initial_response,
|
|
802
|
+
"en_US"
|
|
803
|
+
).encode
|
|
804
|
+
)
|
|
805
|
+
end
|
|
806
|
+
|
|
807
|
+
def send_connection_tune_ok
|
|
808
|
+
@frame_io.write_frame(
|
|
809
|
+
AMQ::Protocol::Connection::TuneOk.encode(
|
|
810
|
+
@negotiated_cmax,
|
|
811
|
+
@negotiated_fm,
|
|
812
|
+
@negotiated_hb
|
|
813
|
+
).encode
|
|
814
|
+
)
|
|
815
|
+
end
|
|
816
|
+
|
|
817
|
+
def send_connection_open
|
|
818
|
+
@frame_io.write_frame(
|
|
819
|
+
AMQ::Protocol::Connection::Open.encode(@vhost).encode
|
|
820
|
+
)
|
|
821
|
+
end
|
|
822
|
+
|
|
823
|
+
def send_connection_close
|
|
824
|
+
# Bound the whole handshake, write included: the broker may never answer,
|
|
825
|
+
# and the write queue may be full behind a stalled socket.
|
|
826
|
+
Async::Task.current.with_timeout(5) do
|
|
827
|
+
@frame_io.write_frame(
|
|
828
|
+
AMQ::Protocol::Connection::Close.encode(200, "Goodbye", 0, 0).encode
|
|
829
|
+
)
|
|
830
|
+
wait_channel0_method(AMQ::Protocol::Connection::CloseOk)
|
|
831
|
+
end
|
|
832
|
+
rescue Async::TimeoutError, ConnectionError, ChannelError, IOError => e
|
|
833
|
+
# Broker didn't answer in time — proceed with the forced close below, which
|
|
834
|
+
# takes the rest of the write queue with it. basic_publish only queues
|
|
835
|
+
# frames, so without confirms this is the publisher's only hint.
|
|
836
|
+
unwritten = @frame_io&.pending_writes.to_i
|
|
837
|
+
return unless unwritten.positive?
|
|
838
|
+
|
|
839
|
+
@logger.warn("Close discarded #{unwritten} queued write(s) (#{e.class})")
|
|
840
|
+
end
|
|
841
|
+
|
|
842
|
+
def negotiate_heartbeat(broker_hb)
|
|
843
|
+
return @heartbeat if broker_hb == 0
|
|
844
|
+
return broker_hb if @heartbeat == 0
|
|
845
|
+
[@heartbeat, broker_hb].min
|
|
846
|
+
end
|
|
847
|
+
|
|
848
|
+
def negotiate_frame_max(broker_fm)
|
|
849
|
+
return @frame_max if broker_fm == 0
|
|
850
|
+
[@frame_max, broker_fm].min
|
|
851
|
+
end
|
|
852
|
+
|
|
853
|
+
# 0 means "no limit" on either side; otherwise the lower value wins.
|
|
854
|
+
def negotiate_channel_max(broker_cmax)
|
|
855
|
+
return (@channel_max == 0 ? 2047 : @channel_max) if broker_cmax == 0
|
|
856
|
+
return broker_cmax if @channel_max == 0
|
|
857
|
+
[@channel_max, broker_cmax].min
|
|
858
|
+
end
|
|
859
|
+
|
|
860
|
+
# The negotiated value T is the heartbeat *timeout*. RabbitMQ and the
|
|
861
|
+
# reference clients send a heartbeat every T/2 and treat the peer as dead
|
|
862
|
+
# after roughly two missed heartbeats, so we send at T/2 as well and
|
|
863
|
+
# declare the broker dead when nothing has arrived for 2×T.
|
|
864
|
+
def start_heartbeat_task
|
|
865
|
+
timeout = @negotiated_hb || 60
|
|
866
|
+
return if timeout == 0
|
|
867
|
+
interval = timeout / 2.0
|
|
868
|
+
dead_after = timeout * 2
|
|
869
|
+
# Seed the timestamp now; the on_frame callback will keep it fresh.
|
|
870
|
+
@last_frame_at = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
871
|
+
|
|
872
|
+
@heartbeat_task = spawn_background do
|
|
873
|
+
loop do
|
|
874
|
+
sleep interval
|
|
875
|
+
break unless open?
|
|
876
|
+
|
|
877
|
+
# Liveness first. The write below needs the socket lock, and after a
|
|
878
|
+
# partition the writer can be parked inside @socket.write holding it
|
|
879
|
+
# with a full send buffer. Checking afterwards meant this task blocked
|
|
880
|
+
# with the rest and nothing declared the peer dead until the kernel
|
|
881
|
+
# gave up retransmitting — around 15 minutes on Linux.
|
|
882
|
+
last = @last_frame_at
|
|
883
|
+
if last
|
|
884
|
+
elapsed = Process.clock_gettime(Process::CLOCK_MONOTONIC) - last
|
|
885
|
+
if elapsed > dead_after
|
|
886
|
+
trigger_recovery(HeartbeatTimeoutError.new("No frame received in #{elapsed.round(1)}s (timeout #{dead_after}s)"))
|
|
887
|
+
break
|
|
888
|
+
end
|
|
889
|
+
end
|
|
890
|
+
|
|
891
|
+
# Bounded so a socket that never drains cannot park this task either.
|
|
892
|
+
# Bounded by the full dead-peer window, not by the send interval: the
|
|
893
|
+
# heartbeat needs the socket lock, and a large batch going out over a
|
|
894
|
+
# slow link can legitimately hold it for longer than one interval.
|
|
895
|
+
begin
|
|
896
|
+
Async::Task.current.with_timeout(dead_after) { @frame_io.write_heartbeat }
|
|
897
|
+
instrument("heartbeat.sent") { { interval: interval } }
|
|
898
|
+
rescue Async::TimeoutError
|
|
899
|
+
trigger_recovery(HeartbeatTimeoutError.new("Heartbeat write blocked for #{dead_after}s — peer is not reading"))
|
|
900
|
+
break
|
|
901
|
+
end
|
|
902
|
+
end
|
|
903
|
+
end
|
|
904
|
+
end
|
|
905
|
+
|
|
906
|
+
# The lowest channel id not in use, so ids come back when channels close.
|
|
907
|
+
# A bare counter ran out at 65535 (the channel field is 16 bits; 65536
|
|
908
|
+
# encodes as channel 0 and the broker drops the connection) and let the
|
|
909
|
+
# process open more channels than channel_max, which the broker also
|
|
910
|
+
# answers by closing the connection. Called with @channel_mutex held.
|
|
911
|
+
def next_channel_id
|
|
912
|
+
@channel_ids ||= fresh_channel_ids
|
|
913
|
+
@channel_ids.allocate or
|
|
914
|
+
raise ChannelLimitError, "channel_max #{@channel_ids.limit} reached: #{@channels.size} channels open on this connection"
|
|
915
|
+
end
|
|
916
|
+
|
|
917
|
+
# The negotiated channel_max; 0 means "no limit", which the protocol caps
|
|
918
|
+
# at 65535 through the width of the frame header's channel field.
|
|
919
|
+
def channel_limit
|
|
920
|
+
max = @negotiated_cmax || @channel_max
|
|
921
|
+
max.nil? || max.zero? ? 65_535 : max
|
|
922
|
+
end
|
|
923
|
+
|
|
924
|
+
# A new allocator sized by the current negotiation, with the ids of the
|
|
925
|
+
# channels we already have marked as taken. Built lazily and again after
|
|
926
|
+
# each handshake, since a reconnect may negotiate a different channel_max
|
|
927
|
+
# while the channels keep their ids.
|
|
928
|
+
def fresh_channel_ids
|
|
929
|
+
ChannelIdAllocator.new(channel_limit).tap do |ids|
|
|
930
|
+
@channels.each_key { |id| ids.reserve(id) }
|
|
931
|
+
end
|
|
932
|
+
end
|
|
933
|
+
|
|
934
|
+
def recover_loop
|
|
935
|
+
delay = @recovery_interval
|
|
936
|
+
attempts = 0
|
|
937
|
+
recovery_started_at = instrument_clock
|
|
938
|
+
|
|
939
|
+
loop do
|
|
940
|
+
break if @closed_by_user
|
|
941
|
+
|
|
942
|
+
attempts += 1
|
|
943
|
+
|
|
944
|
+
# Check retry limit (nil = unlimited)
|
|
945
|
+
if @recovery_attempts && attempts > @recovery_attempts
|
|
946
|
+
@logger.warn("Recovery exhausted after #{@recovery_attempts} attempt(s)")
|
|
947
|
+
@state = :closed
|
|
948
|
+
@recovery_in_progress = false
|
|
949
|
+
exhausted = ConnectionError.new(code: 0, text: "Recovery exhausted after #{@recovery_attempts} attempt(s)")
|
|
950
|
+
@channels.values.each { |ch| ch.mark_closed!(exhausted) rescue nil }
|
|
951
|
+
instrument("recovery.exhausted") { { attempts: @recovery_attempts, reason: :attempts_exceeded } }
|
|
952
|
+
@on_recovery_exhausted&.call(self)
|
|
953
|
+
return
|
|
954
|
+
end
|
|
955
|
+
|
|
956
|
+
jitter = delay * RECOVERY_JITTER * (rand * 2 - 1)
|
|
957
|
+
wait_secs = delay + jitter
|
|
958
|
+
|
|
959
|
+
# Sleep until the delay expires OR session.close signals @recovery_wakeup.
|
|
960
|
+
@recovery_wakeup = Async::Condition.new
|
|
961
|
+
Async::Task.current.with_timeout(wait_secs) do
|
|
962
|
+
@recovery_wakeup.wait
|
|
963
|
+
end rescue nil # TimeoutError is normal; Condition#signal raises nothing
|
|
964
|
+
@recovery_wakeup = nil
|
|
965
|
+
|
|
966
|
+
break if @closed_by_user
|
|
967
|
+
|
|
968
|
+
@logger.info("Recovery attempt #{attempts} (delay was #{delay.round(1)}s)...")
|
|
969
|
+
instrument("recovery.attempt") { { attempt: attempts, delay: wait_secs } }
|
|
970
|
+
@on_recovery_attempt&.call(attempts)
|
|
971
|
+
|
|
972
|
+
connected = false
|
|
973
|
+
shuffled_addresses.each do |target_host, target_port|
|
|
974
|
+
break if @closed_by_user
|
|
975
|
+
begin
|
|
976
|
+
# Wrap the entire reconnect in a timeout so a dead socket doesn't hang forever.
|
|
977
|
+
Async::Task.current.with_timeout(@connect_timeout) do
|
|
978
|
+
raw_socket = open_socket(target_host, target_port)
|
|
979
|
+
@frame_io = build_frame_io(raw_socket)
|
|
980
|
+
@frame_io.start(spawn: method(:spawn_background))
|
|
981
|
+
handshake
|
|
982
|
+
end
|
|
983
|
+
@host = target_host
|
|
984
|
+
@port = target_port
|
|
985
|
+
connected = true
|
|
986
|
+
break
|
|
987
|
+
rescue AuthenticationError => e
|
|
988
|
+
# The credentials no longer work; retrying cannot help.
|
|
989
|
+
@frame_io&.stop rescue nil
|
|
990
|
+
@frame_io = nil
|
|
991
|
+
@logger.error("Recovery abandoned: #{e.message}")
|
|
992
|
+
@state = :closed
|
|
993
|
+
@recovery_in_progress = false
|
|
994
|
+
@channels.values.each { |ch| ch.mark_closed!(e) rescue nil }
|
|
995
|
+
instrument("recovery.exhausted") { { attempts: attempts, reason: :authentication_failed } }
|
|
996
|
+
@on_recovery_exhausted&.call(self)
|
|
997
|
+
@recovery_task = nil
|
|
998
|
+
return
|
|
999
|
+
rescue => e
|
|
1000
|
+
@frame_io&.stop rescue nil
|
|
1001
|
+
@logger.debug("Recovery: #{target_host}:#{target_port} failed — #{e.class}: #{e.message}")
|
|
1002
|
+
end
|
|
1003
|
+
end
|
|
1004
|
+
|
|
1005
|
+
if connected
|
|
1006
|
+
# Only drops from here on belong to this connection: the old socket
|
|
1007
|
+
# goes on failing throughout the backoff, and those errors must not
|
|
1008
|
+
# make the reconnect we just made look like it had dropped too.
|
|
1009
|
+
@recovery_interrupted = false
|
|
1010
|
+
@heartbeat_task&.cancel rescue nil
|
|
1011
|
+
start_heartbeat_task
|
|
1012
|
+
start_channel0_monitor_task
|
|
1013
|
+
@state = :open
|
|
1014
|
+
@channel_ids = fresh_channel_ids
|
|
1015
|
+
# NOTE: keep @recovery_task non-nil until reopen_channels completes so
|
|
1016
|
+
# that session.close can still cancel this task (and therefore interrupt
|
|
1017
|
+
# any wait_for calls inside reopen_after_recovery) if the user closes
|
|
1018
|
+
# the session while channels are being reopened.
|
|
1019
|
+
@logger.info("Recovery successful after #{attempts} attempt(s)")
|
|
1020
|
+
|
|
1021
|
+
# Re-open channels and re-register consumers. @recovery_in_progress
|
|
1022
|
+
# stays true across this: clearing it first leaves a window where a
|
|
1023
|
+
# second drop starts a competing recovery while this one is still
|
|
1024
|
+
# failing channels through mark_closed!, which loses those channels
|
|
1025
|
+
# and their consumers for good.
|
|
1026
|
+
unless reopen_channels
|
|
1027
|
+
# Dropped again mid-reopen. Channels are still :recovering, so go
|
|
1028
|
+
# round again and reopen them on the next connection — but tear
|
|
1029
|
+
# this half-built connection down first. Leaving it up reported the
|
|
1030
|
+
# session as open (so a Cluster would place new channels on a dead
|
|
1031
|
+
# node), held the socket, and leaked a channel-0 monitor per flap.
|
|
1032
|
+
@logger.warn("Connection lost while reopening channels; retrying recovery")
|
|
1033
|
+
@state = :recovering
|
|
1034
|
+
@heartbeat_task&.cancel rescue nil
|
|
1035
|
+
@heartbeat_task = nil
|
|
1036
|
+
@channel0_task&.cancel rescue nil
|
|
1037
|
+
@channel0_task = nil
|
|
1038
|
+
@frame_io&.stop rescue nil
|
|
1039
|
+
delay = [delay * 2, @recovery_max_interval].min
|
|
1040
|
+
next
|
|
1041
|
+
end
|
|
1042
|
+
|
|
1043
|
+
@recovery_in_progress = false
|
|
1044
|
+
instrument("recovery.succeeded") do
|
|
1045
|
+
{ attempts: attempts, host: @host, port: @port, channels: @channels.size,
|
|
1046
|
+
duration: recovery_started_at ? instrument_elapsed(recovery_started_at) : nil }
|
|
1047
|
+
end
|
|
1048
|
+
@on_recovery&.call(self)
|
|
1049
|
+
@recovery_task = nil
|
|
1050
|
+
return
|
|
1051
|
+
else
|
|
1052
|
+
@logger.warn("Recovery attempt #{attempts} failed: no reachable host")
|
|
1053
|
+
delay = [delay * 2, @recovery_max_interval].min
|
|
1054
|
+
break if @closed_by_user
|
|
1055
|
+
end
|
|
1056
|
+
end
|
|
1057
|
+
ensure
|
|
1058
|
+
# Make sure state is consistent if we exit for any reason
|
|
1059
|
+
@recovery_in_progress = false if @closed_by_user
|
|
1060
|
+
end
|
|
1061
|
+
|
|
1062
|
+
# After reconnect: reopen every channel, replay the recorded topology
|
|
1063
|
+
# (session-wide, in dependency order), then release parked callers and
|
|
1064
|
+
# re-register consumers.
|
|
1065
|
+
# Returns false if the connection dropped again while channels were being
|
|
1066
|
+
# reopened. Channels are then left :recovering rather than closed, so the
|
|
1067
|
+
# next pass through recover_loop can reopen them; only a failure that is
|
|
1068
|
+
# this channel's own (a broker rejection) closes it for good.
|
|
1069
|
+
def reopen_channels
|
|
1070
|
+
reopened = []
|
|
1071
|
+
|
|
1072
|
+
@channels.values.each do |channel|
|
|
1073
|
+
begin
|
|
1074
|
+
channel.reopen_on(@frame_io, state: :recovering)
|
|
1075
|
+
reopened << channel
|
|
1076
|
+
rescue => e
|
|
1077
|
+
return false if @recovery_interrupted
|
|
1078
|
+
|
|
1079
|
+
# Fail the channel loudly rather than leave its parked callers hanging.
|
|
1080
|
+
@logger.error("Channel #{channel.channel_id} could not be reopened after recovery: #{e.class}: #{e.message}")
|
|
1081
|
+
channel.mark_closed!(ChannelError.new("Channel could not be reopened after recovery: #{e.message}",
|
|
1082
|
+
channel_id: channel.channel_id))
|
|
1083
|
+
channel_closed(channel.channel_id)
|
|
1084
|
+
end
|
|
1085
|
+
end
|
|
1086
|
+
|
|
1087
|
+
recover_topology_on(reopened) if @recover_topology && !@topology.empty?
|
|
1088
|
+
return false if @recovery_interrupted
|
|
1089
|
+
|
|
1090
|
+
reopened.each { |channel| channel.finish_recovery! rescue nil }
|
|
1091
|
+
true
|
|
1092
|
+
end
|
|
1093
|
+
|
|
1094
|
+
# Re-declare exchanges, then queues, then bindings. Each entity is replayed
|
|
1095
|
+
# on the channel that declared it if that channel is still open, otherwise
|
|
1096
|
+
# on a temporary channel. One entity failing does not stop the others.
|
|
1097
|
+
def recover_topology_on(channels)
|
|
1098
|
+
by_id = channels.to_h { |ch| [ch.channel_id, ch] }
|
|
1099
|
+
temp = nil
|
|
1100
|
+
on = ->(channel_id) { by_id[channel_id] || (temp ||= open_channel) }
|
|
1101
|
+
|
|
1102
|
+
@topology.exchanges.each { |x| on.call(x.channel_id).recover_exchange(x) }
|
|
1103
|
+
@topology.queues.each { |q| on.call(q.channel_id).recover_queue(q) }
|
|
1104
|
+
@topology.queue_bindings.each { |b| on.call(b.channel_id).recover_queue_binding(b) }
|
|
1105
|
+
@topology.exchange_bindings.each { |b| on.call(b.channel_id).recover_exchange_binding(b) }
|
|
1106
|
+
rescue => e
|
|
1107
|
+
@logger.error("Topology recovery aborted: #{e.class}: #{e.message}")
|
|
1108
|
+
ensure
|
|
1109
|
+
temp&.close rescue nil
|
|
1110
|
+
end
|
|
1111
|
+
|
|
1112
|
+
|
|
1113
|
+
# Dedicated long-lived task that drains the channel-0 queue after the
|
|
1114
|
+
# AMQP handshake completes. Handles connection.blocked / connection.unblocked
|
|
1115
|
+
# by delegating to FrameIO's blocked-state gate so write_frame yields
|
|
1116
|
+
# automatically when the broker is resource-constrained.
|
|
1117
|
+
def start_channel0_monitor_task
|
|
1118
|
+
# Cancel any predecessor: a recovery that has to retry starts this again,
|
|
1119
|
+
# and the old one would otherwise sit on a dead queue for good.
|
|
1120
|
+
@channel0_task&.cancel rescue nil
|
|
1121
|
+
@channel0_task = spawn_background { channel0_monitor_loop }
|
|
1122
|
+
end
|
|
1123
|
+
|
|
1124
|
+
def channel0_monitor_loop
|
|
1125
|
+
queue = @frame_io.channel_queue(0)
|
|
1126
|
+
loop do
|
|
1127
|
+
msg = queue.pop
|
|
1128
|
+
break if msg.nil?
|
|
1129
|
+
next unless msg[0] == :method
|
|
1130
|
+
method = msg[1]
|
|
1131
|
+
case method
|
|
1132
|
+
when AMQ::Protocol::Connection::Blocked
|
|
1133
|
+
@frame_io&.set_blocked(method.reason)
|
|
1134
|
+
instrument("connection.blocked") { { reason: method.reason } }
|
|
1135
|
+
@on_blocked&.call(method.reason)
|
|
1136
|
+
when AMQ::Protocol::Connection::Unblocked
|
|
1137
|
+
@frame_io&.set_unblocked
|
|
1138
|
+
instrument("connection.unblocked") { {} }
|
|
1139
|
+
@on_unblocked&.call
|
|
1140
|
+
when AMQ::Protocol::Connection::UpdateSecretOk
|
|
1141
|
+
@update_secret_condition&.signal(method)
|
|
1142
|
+
when AMQ::Protocol::Connection::Close
|
|
1143
|
+
# FrameIO has already answered with close-ok and triggered recovery;
|
|
1144
|
+
# fail a pending update_secret with the broker's reason.
|
|
1145
|
+
@update_secret_condition&.signal(ConnectionError.new(code: method.reply_code, text: method.reply_text))
|
|
1146
|
+
end
|
|
1147
|
+
# Other channel-0 methods during normal operation are intentionally
|
|
1148
|
+
# ignored; the handshake uses wait_channel0_method, not this loop.
|
|
1149
|
+
end
|
|
1150
|
+
rescue => e
|
|
1151
|
+
@logger.debug("Channel-0 monitor exited: #{e.class}: #{e.message}")
|
|
1152
|
+
end
|
|
1153
|
+
end
|
|
1154
|
+
end
|