cable_room 0.8.0.beta1 → 0.8.0.beta2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/cable_room.gemspec +9 -2
- data/exe/cable_room +8 -0
- data/lib/cable_room/bus.rb +342 -0
- data/lib/cable_room/cli.rb +266 -0
- data/lib/cable_room/config.rb +78 -0
- data/lib/cable_room/host/bus_inbound.rb +41 -0
- data/lib/cable_room/host/placement.rb +261 -0
- data/lib/cable_room/host/runner.rb +15 -21
- data/lib/cable_room/host/supervisor.rb +301 -0
- data/lib/cable_room/host.rb +126 -10
- data/lib/cable_room/ports.rb +19 -9
- data/lib/cable_room/railtie.rb +8 -2
- data/lib/cable_room/room/base.rb +27 -24
- data/lib/cable_room/room/host_adapter.rb +9 -4
- data/lib/cable_room/room.rb +3 -1
- data/lib/cable_room/room_member.rb +63 -8
- data/lib/cable_room/version.rb +1 -1
- data/lib/cable_room.rb +68 -0
- metadata +20 -7
- data/lib/cable_room/host/action_cable_inbound.rb +0 -35
|
@@ -0,0 +1,301 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module CableRoom
|
|
4
|
+
class Host
|
|
5
|
+
# The parent process behind `cable_room server --workers N`. It forks N children, runs one
|
|
6
|
+
# block in each, replaces a child that dies, and passes SIGTERM and SIGINT on to every child
|
|
7
|
+
# before exiting itself. It never hosts a room, never subscribes the Bus, and never starts a
|
|
8
|
+
# Host thread: each child builds its own after the fork (`CableRoom.after_fork!`), so nothing
|
|
9
|
+
# with a socket or a thread behind it is ever shared between processes.
|
|
10
|
+
#
|
|
11
|
+
# Supervisor.new(count: 2, logger: logger) { |index| run_a_host(index) }.run
|
|
12
|
+
#
|
|
13
|
+
# The block is the whole child: it runs right after the fork and its return value (an Integer,
|
|
14
|
+
# or nil for 0) is the child's exit status. The child leaves with `exit!`, so `at_exit` hooks
|
|
15
|
+
# the parent registered before forking don't run a second time in every child (Resque does
|
|
16
|
+
# the same for its forked workers). A block that raises is logged and exits 1, which makes
|
|
17
|
+
# the supervisor replace it.
|
|
18
|
+
#
|
|
19
|
+
# Restarts: a child that dies is replaced. One that ran for at least `stable_after` seconds
|
|
20
|
+
# comes back at once; one that died sooner is treated as failing to boot and comes back
|
|
21
|
+
# after a delay that doubles from `first_backoff` up to `max_backoff`. The delay only resets
|
|
22
|
+
# once a replacement stays up `stable_after` seconds, so the fastest any slot can cycle is
|
|
23
|
+
# once per `stable_after` seconds, and a broken app settles at one fork per slot every
|
|
24
|
+
# `max_backoff` seconds rather than thousands a second.
|
|
25
|
+
#
|
|
26
|
+
# Signals: SIGTERM and SIGINT to the parent are relayed to every live child, and the parent
|
|
27
|
+
# waits up to `shutdown_timeout` seconds for them to exit, then SIGKILLs whatever is left.
|
|
28
|
+
# A second SIGTERM or SIGINT while waiting doesn't wait any longer: it SIGKILLs the children
|
|
29
|
+
# at once. Either way the parent exits 0 once they're all gone. A signal sent to one child
|
|
30
|
+
# touches only that child: it exits, and the parent replaces it.
|
|
31
|
+
#
|
|
32
|
+
# The handlers do nothing but write a line to a pipe; the main loop reads the pipe, so no
|
|
33
|
+
# real work runs in signal context. SIGCHLD writes the same pipe to wake the loop when a
|
|
34
|
+
# child exits, and the loop also polls every `POLL_INTERVAL` seconds in case a wake-up is
|
|
35
|
+
# ever missed.
|
|
36
|
+
#
|
|
37
|
+
# If the parent itself dies without relaying anything (SIGKILL, a crash), each child notices
|
|
38
|
+
# through a second pipe (`watch_parent`) and stops as if it had been sent SIGTERM, so a dead
|
|
39
|
+
# supervisor never leaves orphaned workers hosting rooms.
|
|
40
|
+
class Supervisor
|
|
41
|
+
STOP_SIGNALS = %w[TERM INT].freeze
|
|
42
|
+
POLL_INTERVAL = 1
|
|
43
|
+
|
|
44
|
+
Worker = Struct.new(:index, :pid, :started_at)
|
|
45
|
+
|
|
46
|
+
attr_reader :count, :logger
|
|
47
|
+
|
|
48
|
+
def initialize(count:, logger:, stable_after: 5, first_backoff: 1, max_backoff: 30, shutdown_timeout: 25, &body)
|
|
49
|
+
raise ArgumentError, "count must be at least 1 (got #{count.inspect})" unless count.is_a?(Integer) && count >= 1
|
|
50
|
+
raise ArgumentError, "Supervisor needs a block to run in each worker" unless body
|
|
51
|
+
|
|
52
|
+
@count = count
|
|
53
|
+
@logger = logger
|
|
54
|
+
@stable_after = stable_after
|
|
55
|
+
@first_backoff = first_backoff
|
|
56
|
+
@max_backoff = max_backoff
|
|
57
|
+
@shutdown_timeout = shutdown_timeout
|
|
58
|
+
@body = body
|
|
59
|
+
|
|
60
|
+
@workers = {} # index => Worker, for every live child
|
|
61
|
+
@restart_at = {} # index => monotonic time to fork a replacement
|
|
62
|
+
@backoff = {} # index => the delay used for that slot's last boot failure
|
|
63
|
+
@stopping = false
|
|
64
|
+
@kill_at = nil # monotonic time to SIGKILL children that haven't exited
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
# Fork the workers and supervise them until a stop signal has arrived and every child has
|
|
68
|
+
# exited. Blocks. Returns the process exit status (0).
|
|
69
|
+
def run
|
|
70
|
+
@signal_reader, @signal_writer = IO.pipe
|
|
71
|
+
@lifeline_reader, @lifeline_writer = IO.pipe
|
|
72
|
+
previous_traps = install_traps
|
|
73
|
+
|
|
74
|
+
count.times { |index| start_worker(index) }
|
|
75
|
+
|
|
76
|
+
loop do
|
|
77
|
+
wait_for_wakeup
|
|
78
|
+
reap_exited_workers
|
|
79
|
+
break if @stopping && @workers.empty?
|
|
80
|
+
|
|
81
|
+
if @stopping
|
|
82
|
+
kill_stragglers if now >= @kill_at
|
|
83
|
+
else
|
|
84
|
+
start_due_restarts
|
|
85
|
+
end
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
logger.info "all workers exited"
|
|
89
|
+
0
|
|
90
|
+
ensure
|
|
91
|
+
# If we're leaving for any reason other than "every child is gone" (a bug, say), don't
|
|
92
|
+
# orphan the children: tell them to stop too. Then hand the signals back.
|
|
93
|
+
relay("TERM") if @workers.any?
|
|
94
|
+
previous_traps&.each { |sig, handler| trap(sig, handler) }
|
|
95
|
+
[@signal_reader, @signal_writer, @lifeline_reader, @lifeline_writer].each { |io| io&.close }
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
# Ask the supervisor to shut down as if `signal` had arrived: relay it to every worker and
|
|
99
|
+
# exit once they're gone. Safe from any thread and from a trap handler; the work happens on
|
|
100
|
+
# the thread running `run`.
|
|
101
|
+
def stop!(signal = "TERM")
|
|
102
|
+
wake(signal)
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
def stopping?
|
|
106
|
+
@stopping
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
# Pids of the children alive right now.
|
|
110
|
+
def worker_pids
|
|
111
|
+
@workers.values.map(&:pid)
|
|
112
|
+
end
|
|
113
|
+
|
|
114
|
+
private
|
|
115
|
+
|
|
116
|
+
# ---- The parent ----------------------------------------------------------------------
|
|
117
|
+
|
|
118
|
+
def install_traps
|
|
119
|
+
(STOP_SIGNALS + %w[CHLD]).to_h do |sig|
|
|
120
|
+
[sig, trap(sig) { wake(sig) }]
|
|
121
|
+
end
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
# Runs in signal context, so it does the one thing that's safe there: a non-blocking write
|
|
125
|
+
# to the pipe. A full pipe means plenty of wake-ups are already queued, so dropping is fine.
|
|
126
|
+
def wake(signal)
|
|
127
|
+
@signal_writer&.write_nonblock("#{signal}\n")
|
|
128
|
+
rescue IO::WaitWritable, IOError, Errno::EPIPE
|
|
129
|
+
nil
|
|
130
|
+
end
|
|
131
|
+
|
|
132
|
+
# Sleep until a signal or child exit wakes us, a restart or the kill deadline is due, or
|
|
133
|
+
# the poll interval passes. Any stop signal on the pipe starts (or escalates) the shutdown.
|
|
134
|
+
def wait_for_wakeup
|
|
135
|
+
timeout = [POLL_INTERVAL, seconds_until_next_restart, seconds_until_kill].compact.min
|
|
136
|
+
ready, = IO.select([@signal_reader], nil, nil, timeout)
|
|
137
|
+
return unless ready
|
|
138
|
+
|
|
139
|
+
pending = begin
|
|
140
|
+
@signal_reader.read_nonblock(4096)
|
|
141
|
+
rescue IO::WaitReadable, EOFError
|
|
142
|
+
""
|
|
143
|
+
end
|
|
144
|
+
pending.split("\n").each { |signal| begin_stopping(signal) if STOP_SIGNALS.include?(signal) }
|
|
145
|
+
end
|
|
146
|
+
|
|
147
|
+
def begin_stopping(signal)
|
|
148
|
+
if @stopping
|
|
149
|
+
# A second Ctrl-C or TERM: whoever sent it doesn't want to wait for a graceful stop
|
|
150
|
+
logger.warn "got SIG#{signal} again, killing #{@workers.size} worker(s) now"
|
|
151
|
+
relay("KILL")
|
|
152
|
+
@kill_at = Float::INFINITY
|
|
153
|
+
return
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
@stopping = true
|
|
157
|
+
@restart_at.clear
|
|
158
|
+
@kill_at = now + @shutdown_timeout
|
|
159
|
+
logger.info "got SIG#{signal}, relaying to #{@workers.size} worker(s) and waiting up to #{@shutdown_timeout}s for them"
|
|
160
|
+
relay(signal)
|
|
161
|
+
end
|
|
162
|
+
|
|
163
|
+
# Past the shutdown deadline: kill what's left, once (the next reap collects them)
|
|
164
|
+
def kill_stragglers
|
|
165
|
+
logger.warn "#{@workers.size} worker(s) still running after #{@shutdown_timeout}s, killing them"
|
|
166
|
+
relay("KILL")
|
|
167
|
+
@kill_at = Float::INFINITY
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
def relay(signal)
|
|
171
|
+
@workers.each_value do |worker|
|
|
172
|
+
Process.kill(signal, worker.pid)
|
|
173
|
+
rescue Errno::ESRCH
|
|
174
|
+
nil # already gone; the next reap logs it
|
|
175
|
+
end
|
|
176
|
+
end
|
|
177
|
+
|
|
178
|
+
def reap_exited_workers
|
|
179
|
+
@workers.values.each do |worker|
|
|
180
|
+
status = begin
|
|
181
|
+
_, status = Process.wait2(worker.pid, Process::WNOHANG)
|
|
182
|
+
status
|
|
183
|
+
rescue Errno::ECHILD
|
|
184
|
+
:unknown # someone else reaped it; treat it as gone
|
|
185
|
+
end
|
|
186
|
+
next if status.nil?
|
|
187
|
+
|
|
188
|
+
@workers.delete(worker.index)
|
|
189
|
+
uptime = now - worker.started_at
|
|
190
|
+
reason = describe_exit(status)
|
|
191
|
+
|
|
192
|
+
if @stopping
|
|
193
|
+
logger.info "worker #{worker.index} (pid #{worker.pid}) #{reason} after #{uptime.round}s"
|
|
194
|
+
else
|
|
195
|
+
delay = restart_delay(worker.index, uptime)
|
|
196
|
+
@restart_at[worker.index] = now + delay
|
|
197
|
+
logger.warn "worker #{worker.index} (pid #{worker.pid}) #{reason} after #{uptime.round}s; " \
|
|
198
|
+
"restarting #{delay.zero? ? 'now' : "in #{delay}s"}"
|
|
199
|
+
end
|
|
200
|
+
end
|
|
201
|
+
end
|
|
202
|
+
|
|
203
|
+
def describe_exit(status)
|
|
204
|
+
return "exited (status unknown)" if status == :unknown
|
|
205
|
+
return "killed by SIG#{Signal.signame(status.termsig)}" if status.signaled?
|
|
206
|
+
|
|
207
|
+
"exited with status #{status.exitstatus}"
|
|
208
|
+
end
|
|
209
|
+
|
|
210
|
+
# A worker that lived a while and then died gets replaced right away. One that died
|
|
211
|
+
# almost immediately most likely can't boot, so wait, and wait longer each time it happens.
|
|
212
|
+
def restart_delay(index, uptime)
|
|
213
|
+
if uptime >= @stable_after
|
|
214
|
+
@backoff.delete(index)
|
|
215
|
+
0
|
|
216
|
+
else
|
|
217
|
+
@backoff[index] = @backoff[index] ? [@backoff[index] * 2, @max_backoff].min : @first_backoff
|
|
218
|
+
end
|
|
219
|
+
end
|
|
220
|
+
|
|
221
|
+
def start_due_restarts
|
|
222
|
+
due = @restart_at.select { |_, at| at <= now }.keys
|
|
223
|
+
due.each do |index|
|
|
224
|
+
@restart_at.delete(index)
|
|
225
|
+
start_worker(index)
|
|
226
|
+
end
|
|
227
|
+
end
|
|
228
|
+
|
|
229
|
+
def seconds_until_next_restart
|
|
230
|
+
next_at = @restart_at.values.min
|
|
231
|
+
next_at && [next_at - now, 0].max
|
|
232
|
+
end
|
|
233
|
+
|
|
234
|
+
def seconds_until_kill
|
|
235
|
+
@kill_at && [@kill_at - now, 0].max
|
|
236
|
+
end
|
|
237
|
+
|
|
238
|
+
def start_worker(index)
|
|
239
|
+
pid = fork_worker(index)
|
|
240
|
+
@workers[index] = Worker.new(index, pid, now)
|
|
241
|
+
logger.info "forked worker #{index} (pid #{pid})"
|
|
242
|
+
end
|
|
243
|
+
|
|
244
|
+
def now
|
|
245
|
+
Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
246
|
+
end
|
|
247
|
+
|
|
248
|
+
# ---- The child -----------------------------------------------------------------------
|
|
249
|
+
|
|
250
|
+
def fork_worker(index)
|
|
251
|
+
Process.fork do
|
|
252
|
+
status = 1
|
|
253
|
+
begin
|
|
254
|
+
# The child inherits the parent's traps and both ends of its wake-up pipe. Neither
|
|
255
|
+
# belongs here: a relayed SIGTERM must reach the body's own handler (or kill the
|
|
256
|
+
# child by default) instead of writing to a pipe nobody in this process reads.
|
|
257
|
+
(STOP_SIGNALS + %w[CHLD]).each { |sig| trap(sig, "DEFAULT") }
|
|
258
|
+
@signal_reader.close
|
|
259
|
+
@signal_writer.close
|
|
260
|
+
@workers = {}
|
|
261
|
+
|
|
262
|
+
# Give up our copy of the lifeline's write end first, so a sibling can never keep the
|
|
263
|
+
# pipe open after the parent is gone; then watch the read end for the parent's death
|
|
264
|
+
@lifeline_writer.close
|
|
265
|
+
watch_parent(@lifeline_reader)
|
|
266
|
+
|
|
267
|
+
CableRoom.after_fork!
|
|
268
|
+
|
|
269
|
+
result = @body.call(index)
|
|
270
|
+
status = result.is_a?(Integer) ? result : 0
|
|
271
|
+
rescue SystemExit => e
|
|
272
|
+
status = e.status
|
|
273
|
+
rescue SignalException
|
|
274
|
+
raise # let Ruby end the process with the signal, so the parent sees which one
|
|
275
|
+
rescue Exception => e # rubocop:disable Lint/RescueException -- the child must not outlive a broken body
|
|
276
|
+
logger.error "worker #{index} (pid #{Process.pid}) crashed: #{e.class}: #{e.message}\n #{e.backtrace&.first(10)&.join("\n ")}"
|
|
277
|
+
status = 1
|
|
278
|
+
end
|
|
279
|
+
|
|
280
|
+
$stdout.flush
|
|
281
|
+
$stderr.flush
|
|
282
|
+
exit!(status)
|
|
283
|
+
end
|
|
284
|
+
end
|
|
285
|
+
|
|
286
|
+
# Once every child has closed its copy, the parent holds the only write end of the lifeline
|
|
287
|
+
# pipe, so a read here returns EOF exactly when the parent is gone: crashed, or SIGKILLed by
|
|
288
|
+
# the container. A worker must not carry on hosting rooms with nobody supervising it, so it
|
|
289
|
+
# stops itself the same way a relayed SIGTERM would stop it.
|
|
290
|
+
def watch_parent(reader)
|
|
291
|
+
thread = Thread.new do
|
|
292
|
+
reader.read(1)
|
|
293
|
+
Process.kill("TERM", Process.pid)
|
|
294
|
+
end
|
|
295
|
+
thread.name = "cable_room-supervisor-lifeline"
|
|
296
|
+
thread.report_on_exception = false
|
|
297
|
+
thread
|
|
298
|
+
end
|
|
299
|
+
end
|
|
300
|
+
end
|
|
301
|
+
end
|
data/lib/cable_room/host.rb
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
require_relative 'host/worker_pool'
|
|
2
|
-
require_relative 'host/
|
|
2
|
+
require_relative 'host/bus_inbound'
|
|
3
3
|
require_relative 'host/runner'
|
|
4
|
+
require_relative 'host/supervisor'
|
|
5
|
+
require_relative 'host/placement'
|
|
4
6
|
|
|
5
7
|
module CableRoom
|
|
6
8
|
# One Host per process. It owns every Room running here: the worker pool their work runs on,
|
|
@@ -13,21 +15,72 @@ module CableRoom
|
|
|
13
15
|
class Host
|
|
14
16
|
BEAT_INTERVAL = 5.seconds
|
|
15
17
|
|
|
18
|
+
# Raised when a process that doesn't host rooms asks for its Host. With `room_host = :remote`
|
|
19
|
+
# only `cable_room server` hosts rooms; a web process must never start one.
|
|
20
|
+
class NotHosting < StandardError; end
|
|
21
|
+
|
|
22
|
+
NOT_HOSTING_MESSAGE =
|
|
23
|
+
"This process doesn't host rooms: CableRoom.room_host is :remote, so rooms run under " \
|
|
24
|
+
"`cable_room server`. Start rooms there (or call CableRoom::Host.start! first if this " \
|
|
25
|
+
"process really should host them).".freeze
|
|
26
|
+
|
|
27
|
+
@start_mutex = Mutex.new
|
|
28
|
+
|
|
16
29
|
class << self
|
|
30
|
+
# The Host this process runs, or nil when it runs none (a web process in :remote, or an
|
|
31
|
+
# :inline process that hasn't started a room yet). Use this for introspection and cleanup,
|
|
32
|
+
# where "no rooms here" is a normal answer.
|
|
33
|
+
def current
|
|
34
|
+
@current
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
# The Host this process runs. In :inline every process hosts rooms, so the first caller
|
|
38
|
+
# builds it. In :remote only a process that declared itself a rooms host with `start!`
|
|
39
|
+
# (`cable_room server`) has one; anyone else gets NotHosting, so a room can never quietly
|
|
40
|
+
# start in the web tier because some code path reached for the Host.
|
|
17
41
|
def instance
|
|
18
|
-
|
|
42
|
+
current || (CableRoom.inline? ? start! : raise(NotHosting, NOT_HOSTING_MESSAGE))
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
# Make this process a rooms host. Builds the Host once, with `options` (see `new`); later
|
|
46
|
+
# calls return the same one and ignore them. `cable_room server` calls this explicitly; in
|
|
47
|
+
# :inline `instance` calls it for you.
|
|
48
|
+
#
|
|
49
|
+
# In :remote it also starts the Host's Placement, so from then on this process hears
|
|
50
|
+
# `create: true` members' provision requests and races its peers for their rooms. In :inline
|
|
51
|
+
# nobody sends those (a member starts its room itself), so there's nothing to listen for.
|
|
52
|
+
def start!(**options)
|
|
53
|
+
@start_mutex.synchronize do
|
|
54
|
+
@current ||= begin
|
|
55
|
+
host = new(**options)
|
|
56
|
+
host.start_placement if CableRoom.remote?
|
|
57
|
+
host
|
|
58
|
+
rescue StandardError
|
|
59
|
+
host&.shutdown!
|
|
60
|
+
raise
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
# Swap this process's Host for `host` (nil allowed) and return the previous one. Only specs
|
|
66
|
+
# that play both process roles in one Ruby process need this: the web side has to see no
|
|
67
|
+
# Host while another Host runs the rooms.
|
|
68
|
+
def replace_current(host)
|
|
69
|
+
@start_mutex.synchronize { @current.tap { @current = host } }
|
|
19
70
|
end
|
|
20
71
|
end
|
|
21
72
|
|
|
22
|
-
attr_reader :scheduler, :inbound
|
|
73
|
+
attr_reader :scheduler, :inbound, :placement
|
|
23
74
|
|
|
24
75
|
delegate :logger, to: :cable_server
|
|
25
76
|
|
|
26
|
-
# `inbound` is how rooms receive messages
|
|
27
|
-
|
|
77
|
+
# `inbound` is how rooms receive messages: by default the process-wide Bus, waiting up to
|
|
78
|
+
# `subscribe_timeout` seconds for Redis to confirm each room's subscriptions (a room's startup
|
|
79
|
+
# blocks on that). Anything with BusInbound's two methods works.
|
|
80
|
+
def initialize(inbound: nil, subscribe_timeout: Bus::DEFAULT_TIMEOUT)
|
|
28
81
|
@runners = Set.new
|
|
29
82
|
@monitor = Monitor.new
|
|
30
|
-
@inbound = inbound
|
|
83
|
+
@inbound = inbound || BusInbound.new(timeout: subscribe_timeout)
|
|
31
84
|
@scheduler = Rufus::Scheduler.new
|
|
32
85
|
|
|
33
86
|
at_exit do
|
|
@@ -48,10 +101,46 @@ module CableRoom
|
|
|
48
101
|
end
|
|
49
102
|
end
|
|
50
103
|
|
|
51
|
-
# Start
|
|
52
|
-
# Redlock
|
|
53
|
-
|
|
54
|
-
|
|
104
|
+
# Start `room_class`'s room for `key` on this Host if nobody else is running it. Takes the
|
|
105
|
+
# room's Redlock first, so exactly one process wins; returns false when the lock is held.
|
|
106
|
+
# `Room::Base.ensure` calls this on the process Host; `cable_room server` (or a spec playing
|
|
107
|
+
# one) can call it on a Host it owns.
|
|
108
|
+
#
|
|
109
|
+
# `tenant` is the Apartment tenant the room runs in. It defaults to the caller's, which is
|
|
110
|
+
# right for a member's own request thread (:inline). A rooms process has no request tenant,
|
|
111
|
+
# so whatever starts a room there must pass it. The lock key, the room's stream names, and its
|
|
112
|
+
# startup all depend on it, so the whole start runs switched into it.
|
|
113
|
+
def ensure_room(room_class, key = nil, tenant: CableRoom.current_tenant)
|
|
114
|
+
CableRoom.with_tenant(tenant) do
|
|
115
|
+
lock_key = room_class.room_port_key(key)
|
|
116
|
+
# One attempt: a held lock means the room is already running, and retrying (Redlock's
|
|
117
|
+
# default is 3 retries, 200-250 ms apart) would park the caller's ActionCable worker for
|
|
118
|
+
# ~0.7 s on every `create: true` member's ping.
|
|
119
|
+
lock_info = CableRoom.lock_manager.lock(lock_key, room_class::LOCK_DURATION.in_milliseconds, retry_count: 0)
|
|
120
|
+
next false unless lock_info
|
|
121
|
+
|
|
122
|
+
begin
|
|
123
|
+
start_room(
|
|
124
|
+
room_class,
|
|
125
|
+
key,
|
|
126
|
+
lock_info,
|
|
127
|
+
watchdog_interval: room_class::WATCH_DOG_INTERVAL,
|
|
128
|
+
lock_duration: room_class::LOCK_DURATION,
|
|
129
|
+
tenant: tenant,
|
|
130
|
+
)
|
|
131
|
+
rescue => e
|
|
132
|
+
CableRoom.lock_manager.unlock(lock_info)
|
|
133
|
+
raise e
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
true
|
|
137
|
+
end
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
# Start running a room here. The caller (`ensure_room`) has already taken the room's Redlock;
|
|
141
|
+
# the runner renews it from now on and releases it when the room dies.
|
|
142
|
+
def start_room(room_class, key, lock_info, watchdog_interval:, lock_duration:, tenant: CableRoom.current_tenant)
|
|
143
|
+
runner = Runner.new(self, room_class, key, lock_info, watchdog_interval:, lock_duration:, tenant:)
|
|
55
144
|
runner.start!
|
|
56
145
|
runner
|
|
57
146
|
end
|
|
@@ -95,6 +184,22 @@ module CableRoom
|
|
|
95
184
|
@shutdown
|
|
96
185
|
end
|
|
97
186
|
|
|
187
|
+
# Start hearing provision requests (see Host::Placement). `Host.start!` does this for a :remote
|
|
188
|
+
# rooms process; a Host built with `new` (a spec playing one) calls it itself. `options` go to
|
|
189
|
+
# Placement.new the first time; later calls restart the same Placement.
|
|
190
|
+
def start_placement(**options)
|
|
191
|
+
placement = @monitor.synchronize do
|
|
192
|
+
raise "Cannot start placement after shutdown" if @shutdown
|
|
193
|
+
@placement ||= Placement.new(self, **options)
|
|
194
|
+
end
|
|
195
|
+
placement.start
|
|
196
|
+
end
|
|
197
|
+
|
|
198
|
+
# Stop hearing requests and drop any claim still waiting. Nothing starts a room here after this.
|
|
199
|
+
def stop_placement
|
|
200
|
+
@placement&.stop
|
|
201
|
+
end
|
|
202
|
+
|
|
98
203
|
# Stop everything: the beat, every room (gracefully, with a bounded wait), then the pool.
|
|
99
204
|
# Runs once; later calls do nothing.
|
|
100
205
|
def shutdown!
|
|
@@ -103,6 +208,8 @@ module CableRoom
|
|
|
103
208
|
@shutdown = true
|
|
104
209
|
end
|
|
105
210
|
|
|
211
|
+
# Stop claiming first: a room started after this point would only be shut down again
|
|
212
|
+
stop_placement
|
|
106
213
|
scheduler.shutdown
|
|
107
214
|
shutdown_rooms!
|
|
108
215
|
worker_pool.executor.shutdown
|
|
@@ -110,6 +217,15 @@ module CableRoom
|
|
|
110
217
|
worker_pool.executor.kill
|
|
111
218
|
end
|
|
112
219
|
|
|
220
|
+
# Give this Host up without touching anything it holds. For a copy of a Host that a forked
|
|
221
|
+
# child inherited from its parent (see `CableRoom.after_fork!`): its rooms, locks, and threads
|
|
222
|
+
# belong to the parent, so the child must neither close those rooms nor let the `at_exit`
|
|
223
|
+
# hook above do it when the child exits. Marks it shut down so both are no-ops.
|
|
224
|
+
def abandon!
|
|
225
|
+
@shutdown = true
|
|
226
|
+
self
|
|
227
|
+
end
|
|
228
|
+
|
|
113
229
|
# Ask every room to finish what it has queued and shut down, then give them up to
|
|
114
230
|
# `wait` seconds to do it
|
|
115
231
|
def shutdown_rooms!(wait: 15)
|
data/lib/cable_room/ports.rb
CHANGED
|
@@ -1,4 +1,17 @@
|
|
|
1
1
|
module CableRoom
|
|
2
|
+
# The `ports[...]` DSL shared by a Room and by a member's RoomMembership. A port is a named
|
|
3
|
+
# lane between a room and its members; `ports[:x] << msg` sends on it and `ports[:x].stream`
|
|
4
|
+
# listens on it. Which transport that touches depends on which side you're on, and each side
|
|
5
|
+
# implements the three methods below:
|
|
6
|
+
#
|
|
7
|
+
# * Member side (RoomMembership): sending publishes on the room's Bus channel for the port
|
|
8
|
+
# (member→room); listening is an ActionCable stream (room→member).
|
|
9
|
+
# * Room side (Room::Base and Room::HostAdapter): sending is an ActionCable broadcast on the
|
|
10
|
+
# port's stream; listening subscribes on the Bus through the room's Host.
|
|
11
|
+
#
|
|
12
|
+
# So a member's `ports[:custom] << msg` and a room's `ports[:custom].stream` meet on the Bus,
|
|
13
|
+
# and a room's `ports[:custom] << msg` and a member's `ports[:custom].stream` meet on the
|
|
14
|
+
# ActionCable stream named by `room_port_key`. Neither side needs to know that.
|
|
2
15
|
module Ports
|
|
3
16
|
extend ActiveSupport::Concern
|
|
4
17
|
|
|
@@ -6,22 +19,19 @@ module CableRoom
|
|
|
6
19
|
@ports_proxy ||= PortsProxy.new(self)
|
|
7
20
|
end
|
|
8
21
|
|
|
9
|
-
#
|
|
10
|
-
#
|
|
22
|
+
# Listen on `port`, handing each decoded message to the block. Ports streamed with
|
|
23
|
+
# `auto_close: true` are stopped by `close_streamed_ports!`.
|
|
11
24
|
def stream_port(port, auto_close: true, &blk)
|
|
12
|
-
|
|
13
|
-
_streamed_ports << port if auto_close
|
|
25
|
+
raise NotImplementedError, "#{self.class} must implement stream_port"
|
|
14
26
|
end
|
|
15
27
|
|
|
16
28
|
def close_streamed_ports!
|
|
17
|
-
|
|
18
|
-
@cable_channel.stop_stream_from(room_port_key(port))
|
|
19
|
-
end
|
|
20
|
-
_streamed_ports.clear
|
|
29
|
+
raise NotImplementedError, "#{self.class} must implement close_streamed_ports!"
|
|
21
30
|
end
|
|
22
31
|
|
|
32
|
+
# Send `data` on `port`.
|
|
23
33
|
def port_transmit(port, data)
|
|
24
|
-
|
|
34
|
+
raise NotImplementedError, "#{self.class} must implement port_transmit"
|
|
25
35
|
end
|
|
26
36
|
|
|
27
37
|
protected
|
data/lib/cable_room/railtie.rb
CHANGED
|
@@ -14,11 +14,17 @@ module CableRoom
|
|
|
14
14
|
runner do
|
|
15
15
|
end
|
|
16
16
|
|
|
17
|
+
# Read CABLE_ROOM_HOST now so a bad value fails the boot, not the first room
|
|
18
|
+
initializer "cable_room.room_host" do
|
|
19
|
+
CableRoom.room_host
|
|
20
|
+
end
|
|
21
|
+
|
|
17
22
|
initializer "cable_room.hook_action_cable_restart" do
|
|
18
23
|
module ActionCableServerExtensions
|
|
19
|
-
# ActionCable restarts when the app reloads; stop every room with it
|
|
24
|
+
# ActionCable restarts when the app reloads; stop every room with it. A process that
|
|
25
|
+
# hosts no rooms (a web process in :remote) has nothing to stop.
|
|
20
26
|
def restart
|
|
21
|
-
CableRoom::Host.
|
|
27
|
+
CableRoom::Host.current&.stop_all_rooms!
|
|
22
28
|
super
|
|
23
29
|
end
|
|
24
30
|
end
|
data/lib/cable_room/room/base.rb
CHANGED
|
@@ -10,26 +10,13 @@ module CableRoom
|
|
|
10
10
|
class << self
|
|
11
11
|
# Start this room in the current process if nobody else is running it. Takes the room's
|
|
12
12
|
# Redlock first, so exactly one process wins; returns false when the lock is held.
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
Host.instance.start_room(
|
|
22
|
-
self,
|
|
23
|
-
key,
|
|
24
|
-
lock_info,
|
|
25
|
-
watchdog_interval: self::WATCH_DOG_INTERVAL,
|
|
26
|
-
lock_duration: self::LOCK_DURATION,
|
|
27
|
-
)
|
|
28
|
-
|
|
29
|
-
true
|
|
30
|
-
rescue => e
|
|
31
|
-
CableRoom.lock_manager.unlock(lock_info) if lock_info
|
|
32
|
-
raise e
|
|
13
|
+
# `tenant` defaults to the caller's (see Host#ensure_room).
|
|
14
|
+
#
|
|
15
|
+
# Only a process that hosts rooms can do this: every process in :inline, and only
|
|
16
|
+
# `cable_room server` in :remote. Anywhere else `Host.instance` raises Host::NotHosting
|
|
17
|
+
# rather than letting a room start where it doesn't belong.
|
|
18
|
+
def ensure(key = nil, tenant: CableRoom.current_tenant)
|
|
19
|
+
Host.instance.ensure_room(self, key, tenant: tenant)
|
|
33
20
|
end
|
|
34
21
|
|
|
35
22
|
# The stream name for a room, or for one of its ports: "RoomClass:key" or
|
|
@@ -42,14 +29,23 @@ module CableRoom
|
|
|
42
29
|
stream_namer.broadcasting_for(full_key)
|
|
43
30
|
end
|
|
44
31
|
|
|
32
|
+
# The Bus channel members publish on to reach a room's inbound `port`, and the one the
|
|
33
|
+
# room's Host subscribes to for it: `cr:{room_port_key}:in` for the main port and
|
|
34
|
+
# `cr:{room_port_key}:in:{port}` for a custom one (see Bus.inbound_channel).
|
|
35
|
+
def inbound_channel(room_key, port = ROOM_IN_CHANNEL)
|
|
36
|
+
port = port.to_s == ROOM_IN_CHANNEL.to_s ? nil : port.to_s
|
|
37
|
+
Bus.inbound_channel(self, room_key, port)
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
# Send a message to a room from anywhere in the app: the same Bus channel a member uses
|
|
45
41
|
def send_message(room_key, data, port: ROOM_IN_CHANNEL)
|
|
46
|
-
|
|
42
|
+
CableRoom.bus.publish(inbound_channel(room_key, port), data)
|
|
47
43
|
end
|
|
48
44
|
|
|
45
|
+
# Every instance of this room class running in this process. Empty when the process
|
|
46
|
+
# hosts no rooms (a web process in :remote).
|
|
49
47
|
def locally_running_instances
|
|
50
|
-
|
|
51
|
-
runner.room_class == self
|
|
52
|
-
end.map(&:room)
|
|
48
|
+
Room.locally_open_rooms.select { |room| room.class == self }
|
|
53
49
|
end
|
|
54
50
|
|
|
55
51
|
private
|
|
@@ -97,6 +93,13 @@ module CableRoom
|
|
|
97
93
|
port_transmit(ROOM_OUT_CHANNEL, data)
|
|
98
94
|
end
|
|
99
95
|
|
|
96
|
+
# The room's sending side of Ports: room→member messages go out as an ActionCable
|
|
97
|
+
# broadcast on the stream named by `room_port_key(port)`. (The listening side is in
|
|
98
|
+
# Room::HostAdapter.) Public because `ports[:x] << msg` reaches it through a PortProxy.
|
|
99
|
+
def port_transmit(port, data)
|
|
100
|
+
ActionCable.server.broadcast(room_port_key(port), data)
|
|
101
|
+
end
|
|
102
|
+
|
|
100
103
|
protected
|
|
101
104
|
|
|
102
105
|
def room_class
|
|
@@ -23,22 +23,27 @@ module CableRoom
|
|
|
23
23
|
end
|
|
24
24
|
end
|
|
25
25
|
|
|
26
|
-
# The room side of Ports
|
|
27
|
-
# `on_live` runs once the
|
|
26
|
+
# The room's listening side of Ports: subscribe, through the Host, to the Bus channel
|
|
27
|
+
# members publish on for `port` (see Room::Base.inbound_channel). `on_live` runs once the
|
|
28
|
+
# subscription is confirmed.
|
|
28
29
|
def stream_port(port, auto_close: true, on_live: nil, &blk)
|
|
29
|
-
@runner.subscribe(
|
|
30
|
+
@runner.subscribe(inbound_channel(port), on_live: on_live, &blk)
|
|
30
31
|
_streamed_ports << port if auto_close
|
|
31
32
|
end
|
|
32
33
|
|
|
33
34
|
def close_streamed_ports!
|
|
34
35
|
_streamed_ports.each do |port|
|
|
35
|
-
@runner.unsubscribe(
|
|
36
|
+
@runner.unsubscribe(inbound_channel(port))
|
|
36
37
|
end
|
|
37
38
|
_streamed_ports.clear
|
|
38
39
|
end
|
|
39
40
|
|
|
40
41
|
protected
|
|
41
42
|
|
|
43
|
+
def inbound_channel(port)
|
|
44
|
+
self.class.inbound_channel(key, port)
|
|
45
|
+
end
|
|
46
|
+
|
|
42
47
|
def start_periodic_timer(callback, every:)
|
|
43
48
|
@runner.start_periodic_timer(-> { instance_exec(&callback) }, every: every)
|
|
44
49
|
end
|
data/lib/cable_room/room.rb
CHANGED
|
@@ -20,8 +20,10 @@ module CableRoom
|
|
|
20
20
|
autoload :Broadcasting
|
|
21
21
|
end
|
|
22
22
|
|
|
23
|
+
# Every room running in this process. Asking never creates a Host, so a process that hosts
|
|
24
|
+
# no rooms (a web process in :remote) gets an empty list.
|
|
23
25
|
def self.locally_open_rooms
|
|
24
|
-
Host.
|
|
26
|
+
Host.current&.rooms || []
|
|
25
27
|
end
|
|
26
28
|
end
|
|
27
29
|
end
|