cable_room 0.8.0.beta1 → 0.8.0.beta2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,301 @@
1
+ # frozen_string_literal: true
2
+
3
+ module CableRoom
4
+ class Host
5
+ # The parent process behind `cable_room server --workers N`. It forks N children, runs one
6
+ # block in each, replaces a child that dies, and passes SIGTERM and SIGINT on to every child
7
+ # before exiting itself. It never hosts a room, never subscribes the Bus, and never starts a
8
+ # Host thread: each child builds its own after the fork (`CableRoom.after_fork!`), so nothing
9
+ # with a socket or a thread behind it is ever shared between processes.
10
+ #
11
+ # Supervisor.new(count: 2, logger: logger) { |index| run_a_host(index) }.run
12
+ #
13
+ # The block is the whole child: it runs right after the fork and its return value (an Integer,
14
+ # or nil for 0) is the child's exit status. The child leaves with `exit!`, so `at_exit` hooks
15
+ # the parent registered before forking don't run a second time in every child (Resque does
16
+ # the same for its forked workers). A block that raises is logged and exits 1, which makes
17
+ # the supervisor replace it.
18
+ #
19
+ # Restarts: a child that dies is replaced. One that ran for at least `stable_after` seconds
20
+ # comes back at once; one that died sooner is treated as failing to boot and comes back
21
+ # after a delay that doubles from `first_backoff` up to `max_backoff`. The delay only resets
22
+ # once a replacement stays up `stable_after` seconds, so the fastest any slot can cycle is
23
+ # once per `stable_after` seconds, and a broken app settles at one fork per slot every
24
+ # `max_backoff` seconds rather than thousands a second.
25
+ #
26
+ # Signals: SIGTERM and SIGINT to the parent are relayed to every live child, and the parent
27
+ # waits up to `shutdown_timeout` seconds for them to exit, then SIGKILLs whatever is left.
28
+ # A second SIGTERM or SIGINT while waiting doesn't wait any longer: it SIGKILLs the children
29
+ # at once. Either way the parent exits 0 once they're all gone. A signal sent to one child
30
+ # touches only that child: it exits, and the parent replaces it.
31
+ #
32
+ # The handlers do nothing but write a line to a pipe; the main loop reads the pipe, so no
33
+ # real work runs in signal context. SIGCHLD writes the same pipe to wake the loop when a
34
+ # child exits, and the loop also polls every `POLL_INTERVAL` seconds in case a wake-up is
35
+ # ever missed.
36
+ #
37
+ # If the parent itself dies without relaying anything (SIGKILL, a crash), each child notices
38
+ # through a second pipe (`watch_parent`) and stops as if it had been sent SIGTERM, so a dead
39
+ # supervisor never leaves orphaned workers hosting rooms.
40
+ class Supervisor
41
+ STOP_SIGNALS = %w[TERM INT].freeze
42
+ POLL_INTERVAL = 1
43
+
44
+ Worker = Struct.new(:index, :pid, :started_at)
45
+
46
+ attr_reader :count, :logger
47
+
48
+ def initialize(count:, logger:, stable_after: 5, first_backoff: 1, max_backoff: 30, shutdown_timeout: 25, &body)
49
+ raise ArgumentError, "count must be at least 1 (got #{count.inspect})" unless count.is_a?(Integer) && count >= 1
50
+ raise ArgumentError, "Supervisor needs a block to run in each worker" unless body
51
+
52
+ @count = count
53
+ @logger = logger
54
+ @stable_after = stable_after
55
+ @first_backoff = first_backoff
56
+ @max_backoff = max_backoff
57
+ @shutdown_timeout = shutdown_timeout
58
+ @body = body
59
+
60
+ @workers = {} # index => Worker, for every live child
61
+ @restart_at = {} # index => monotonic time to fork a replacement
62
+ @backoff = {} # index => the delay used for that slot's last boot failure
63
+ @stopping = false
64
+ @kill_at = nil # monotonic time to SIGKILL children that haven't exited
65
+ end
66
+
67
+ # Fork the workers and supervise them until a stop signal has arrived and every child has
68
+ # exited. Blocks. Returns the process exit status (0).
69
+ def run
70
+ @signal_reader, @signal_writer = IO.pipe
71
+ @lifeline_reader, @lifeline_writer = IO.pipe
72
+ previous_traps = install_traps
73
+
74
+ count.times { |index| start_worker(index) }
75
+
76
+ loop do
77
+ wait_for_wakeup
78
+ reap_exited_workers
79
+ break if @stopping && @workers.empty?
80
+
81
+ if @stopping
82
+ kill_stragglers if now >= @kill_at
83
+ else
84
+ start_due_restarts
85
+ end
86
+ end
87
+
88
+ logger.info "all workers exited"
89
+ 0
90
+ ensure
91
+ # If we're leaving for any reason other than "every child is gone" (a bug, say), don't
92
+ # orphan the children: tell them to stop too. Then hand the signals back.
93
+ relay("TERM") if @workers.any?
94
+ previous_traps&.each { |sig, handler| trap(sig, handler) }
95
+ [@signal_reader, @signal_writer, @lifeline_reader, @lifeline_writer].each { |io| io&.close }
96
+ end
97
+
98
+ # Ask the supervisor to shut down as if `signal` had arrived: relay it to every worker and
99
+ # exit once they're gone. Safe from any thread and from a trap handler; the work happens on
100
+ # the thread running `run`.
101
+ def stop!(signal = "TERM")
102
+ wake(signal)
103
+ end
104
+
105
+ def stopping?
106
+ @stopping
107
+ end
108
+
109
+ # Pids of the children alive right now.
110
+ def worker_pids
111
+ @workers.values.map(&:pid)
112
+ end
113
+
114
+ private
115
+
116
+ # ---- The parent ----------------------------------------------------------------------
117
+
118
+ def install_traps
119
+ (STOP_SIGNALS + %w[CHLD]).to_h do |sig|
120
+ [sig, trap(sig) { wake(sig) }]
121
+ end
122
+ end
123
+
124
+ # Runs in signal context, so it does the one thing that's safe there: a non-blocking write
125
+ # to the pipe. A full pipe means plenty of wake-ups are already queued, so dropping is fine.
126
+ def wake(signal)
127
+ @signal_writer&.write_nonblock("#{signal}\n")
128
+ rescue IO::WaitWritable, IOError, Errno::EPIPE
129
+ nil
130
+ end
131
+
132
+ # Sleep until a signal or child exit wakes us, a restart or the kill deadline is due, or
133
+ # the poll interval passes. Any stop signal on the pipe starts (or escalates) the shutdown.
134
+ def wait_for_wakeup
135
+ timeout = [POLL_INTERVAL, seconds_until_next_restart, seconds_until_kill].compact.min
136
+ ready, = IO.select([@signal_reader], nil, nil, timeout)
137
+ return unless ready
138
+
139
+ pending = begin
140
+ @signal_reader.read_nonblock(4096)
141
+ rescue IO::WaitReadable, EOFError
142
+ ""
143
+ end
144
+ pending.split("\n").each { |signal| begin_stopping(signal) if STOP_SIGNALS.include?(signal) }
145
+ end
146
+
147
+ def begin_stopping(signal)
148
+ if @stopping
149
+ # A second Ctrl-C or TERM: whoever sent it doesn't want to wait for a graceful stop
150
+ logger.warn "got SIG#{signal} again, killing #{@workers.size} worker(s) now"
151
+ relay("KILL")
152
+ @kill_at = Float::INFINITY
153
+ return
154
+ end
155
+
156
+ @stopping = true
157
+ @restart_at.clear
158
+ @kill_at = now + @shutdown_timeout
159
+ logger.info "got SIG#{signal}, relaying to #{@workers.size} worker(s) and waiting up to #{@shutdown_timeout}s for them"
160
+ relay(signal)
161
+ end
162
+
163
+ # Past the shutdown deadline: kill what's left, once (the next reap collects them)
164
+ def kill_stragglers
165
+ logger.warn "#{@workers.size} worker(s) still running after #{@shutdown_timeout}s, killing them"
166
+ relay("KILL")
167
+ @kill_at = Float::INFINITY
168
+ end
169
+
170
+ def relay(signal)
171
+ @workers.each_value do |worker|
172
+ Process.kill(signal, worker.pid)
173
+ rescue Errno::ESRCH
174
+ nil # already gone; the next reap logs it
175
+ end
176
+ end
177
+
178
+ def reap_exited_workers
179
+ @workers.values.each do |worker|
180
+ status = begin
181
+ _, status = Process.wait2(worker.pid, Process::WNOHANG)
182
+ status
183
+ rescue Errno::ECHILD
184
+ :unknown # someone else reaped it; treat it as gone
185
+ end
186
+ next if status.nil?
187
+
188
+ @workers.delete(worker.index)
189
+ uptime = now - worker.started_at
190
+ reason = describe_exit(status)
191
+
192
+ if @stopping
193
+ logger.info "worker #{worker.index} (pid #{worker.pid}) #{reason} after #{uptime.round}s"
194
+ else
195
+ delay = restart_delay(worker.index, uptime)
196
+ @restart_at[worker.index] = now + delay
197
+ logger.warn "worker #{worker.index} (pid #{worker.pid}) #{reason} after #{uptime.round}s; " \
198
+ "restarting #{delay.zero? ? 'now' : "in #{delay}s"}"
199
+ end
200
+ end
201
+ end
202
+
203
+ def describe_exit(status)
204
+ return "exited (status unknown)" if status == :unknown
205
+ return "killed by SIG#{Signal.signame(status.termsig)}" if status.signaled?
206
+
207
+ "exited with status #{status.exitstatus}"
208
+ end
209
+
210
+ # A worker that lived a while and then died gets replaced right away. One that died
211
+ # almost immediately most likely can't boot, so wait, and wait longer each time it happens.
212
+ def restart_delay(index, uptime)
213
+ if uptime >= @stable_after
214
+ @backoff.delete(index)
215
+ 0
216
+ else
217
+ @backoff[index] = @backoff[index] ? [@backoff[index] * 2, @max_backoff].min : @first_backoff
218
+ end
219
+ end
220
+
221
+ def start_due_restarts
222
+ due = @restart_at.select { |_, at| at <= now }.keys
223
+ due.each do |index|
224
+ @restart_at.delete(index)
225
+ start_worker(index)
226
+ end
227
+ end
228
+
229
+ def seconds_until_next_restart
230
+ next_at = @restart_at.values.min
231
+ next_at && [next_at - now, 0].max
232
+ end
233
+
234
+ def seconds_until_kill
235
+ @kill_at && [@kill_at - now, 0].max
236
+ end
237
+
238
+ def start_worker(index)
239
+ pid = fork_worker(index)
240
+ @workers[index] = Worker.new(index, pid, now)
241
+ logger.info "forked worker #{index} (pid #{pid})"
242
+ end
243
+
244
+ def now
245
+ Process.clock_gettime(Process::CLOCK_MONOTONIC)
246
+ end
247
+
248
+ # ---- The child -----------------------------------------------------------------------
249
+
250
+ def fork_worker(index)
251
+ Process.fork do
252
+ status = 1
253
+ begin
254
+ # The child inherits the parent's traps and both ends of its wake-up pipe. Neither
255
+ # belongs here: a relayed SIGTERM must reach the body's own handler (or kill the
256
+ # child by default) instead of writing to a pipe nobody in this process reads.
257
+ (STOP_SIGNALS + %w[CHLD]).each { |sig| trap(sig, "DEFAULT") }
258
+ @signal_reader.close
259
+ @signal_writer.close
260
+ @workers = {}
261
+
262
+ # Give up our copy of the lifeline's write end first, so a sibling can never keep the
263
+ # pipe open after the parent is gone; then watch the read end for the parent's death
264
+ @lifeline_writer.close
265
+ watch_parent(@lifeline_reader)
266
+
267
+ CableRoom.after_fork!
268
+
269
+ result = @body.call(index)
270
+ status = result.is_a?(Integer) ? result : 0
271
+ rescue SystemExit => e
272
+ status = e.status
273
+ rescue SignalException
274
+ raise # let Ruby end the process with the signal, so the parent sees which one
275
+ rescue Exception => e # rubocop:disable Lint/RescueException -- the child must not outlive a broken body
276
+ logger.error "worker #{index} (pid #{Process.pid}) crashed: #{e.class}: #{e.message}\n #{e.backtrace&.first(10)&.join("\n ")}"
277
+ status = 1
278
+ end
279
+
280
+ $stdout.flush
281
+ $stderr.flush
282
+ exit!(status)
283
+ end
284
+ end
285
+
286
+ # Once every child has closed its copy, the parent holds the only write end of the lifeline
287
+ # pipe, so a read here returns EOF exactly when the parent is gone: crashed, or SIGKILLed by
288
+ # the container. A worker must not carry on hosting rooms with nobody supervising it, so it
289
+ # stops itself the same way a relayed SIGTERM would stop it.
290
+ def watch_parent(reader)
291
+ thread = Thread.new do
292
+ reader.read(1)
293
+ Process.kill("TERM", Process.pid)
294
+ end
295
+ thread.name = "cable_room-supervisor-lifeline"
296
+ thread.report_on_exception = false
297
+ thread
298
+ end
299
+ end
300
+ end
301
+ end
@@ -1,6 +1,8 @@
1
1
  require_relative 'host/worker_pool'
2
- require_relative 'host/action_cable_inbound'
2
+ require_relative 'host/bus_inbound'
3
3
  require_relative 'host/runner'
4
+ require_relative 'host/supervisor'
5
+ require_relative 'host/placement'
4
6
 
5
7
  module CableRoom
6
8
  # One Host per process. It owns every Room running here: the worker pool their work runs on,
@@ -13,21 +15,72 @@ module CableRoom
13
15
  class Host
14
16
  BEAT_INTERVAL = 5.seconds
15
17
 
18
+ # Raised when a process that doesn't host rooms asks for its Host. With `room_host = :remote`
19
+ # only `cable_room server` hosts rooms; a web process must never start one.
20
+ class NotHosting < StandardError; end
21
+
22
+ NOT_HOSTING_MESSAGE =
23
+ "This process doesn't host rooms: CableRoom.room_host is :remote, so rooms run under " \
24
+ "`cable_room server`. Start rooms there (or call CableRoom::Host.start! first if this " \
25
+ "process really should host them).".freeze
26
+
27
+ @start_mutex = Mutex.new
28
+
16
29
  class << self
30
+ # The Host this process runs, or nil when it runs none (a web process in :remote, or an
31
+ # :inline process that hasn't started a room yet). Use this for introspection and cleanup,
32
+ # where "no rooms here" is a normal answer.
33
+ def current
34
+ @current
35
+ end
36
+
37
+ # The Host this process runs. In :inline every process hosts rooms, so the first caller
38
+ # builds it. In :remote only a process that declared itself a rooms host with `start!`
39
+ # (`cable_room server`) has one; anyone else gets NotHosting, so a room can never quietly
40
+ # start in the web tier because some code path reached for the Host.
17
41
  def instance
18
- @instance ||= new
42
+ current || (CableRoom.inline? ? start! : raise(NotHosting, NOT_HOSTING_MESSAGE))
43
+ end
44
+
45
+ # Make this process a rooms host. Builds the Host once, with `options` (see `new`); later
46
+ # calls return the same one and ignore them. `cable_room server` calls this explicitly; in
47
+ # :inline `instance` calls it for you.
48
+ #
49
+ # In :remote it also starts the Host's Placement, so from then on this process hears
50
+ # `create: true` members' provision requests and races its peers for their rooms. In :inline
51
+ # nobody sends those (a member starts its room itself), so there's nothing to listen for.
52
+ def start!(**options)
53
+ @start_mutex.synchronize do
54
+ @current ||= begin
55
+ host = new(**options)
56
+ host.start_placement if CableRoom.remote?
57
+ host
58
+ rescue StandardError
59
+ host&.shutdown!
60
+ raise
61
+ end
62
+ end
63
+ end
64
+
65
+ # Swap this process's Host for `host` (nil allowed) and return the previous one. Only specs
66
+ # that play both process roles in one Ruby process need this: the web side has to see no
67
+ # Host while another Host runs the rooms.
68
+ def replace_current(host)
69
+ @start_mutex.synchronize { @current.tap { @current = host } }
19
70
  end
20
71
  end
21
72
 
22
- attr_reader :scheduler, :inbound
73
+ attr_reader :scheduler, :inbound, :placement
23
74
 
24
75
  delegate :logger, to: :cable_server
25
76
 
26
- # `inbound` is how rooms receive messages. Anything with ActionCableInbound's two methods works.
27
- def initialize(inbound: ActionCableInbound.new)
77
+ # `inbound` is how rooms receive messages: by default the process-wide Bus, waiting up to
78
+ # `subscribe_timeout` seconds for Redis to confirm each room's subscriptions (a room's startup
79
+ # blocks on that). Anything with BusInbound's two methods works.
80
+ def initialize(inbound: nil, subscribe_timeout: Bus::DEFAULT_TIMEOUT)
28
81
  @runners = Set.new
29
82
  @monitor = Monitor.new
30
- @inbound = inbound
83
+ @inbound = inbound || BusInbound.new(timeout: subscribe_timeout)
31
84
  @scheduler = Rufus::Scheduler.new
32
85
 
33
86
  at_exit do
@@ -48,10 +101,46 @@ module CableRoom
48
101
  end
49
102
  end
50
103
 
51
- # Start running a room here. The caller (Room::Base.ensure) has already taken the room's
52
- # Redlock; the runner renews it from now on and releases it when the room dies.
53
- def start_room(room_class, key, lock_info, watchdog_interval:, lock_duration:)
54
- runner = Runner.new(self, room_class, key, lock_info, watchdog_interval:, lock_duration:)
104
+ # Start `room_class`'s room for `key` on this Host if nobody else is running it. Takes the
105
+ # room's Redlock first, so exactly one process wins; returns false when the lock is held.
106
+ # `Room::Base.ensure` calls this on the process Host; `cable_room server` (or a spec playing
107
+ # one) can call it on a Host it owns.
108
+ #
109
+ # `tenant` is the Apartment tenant the room runs in. It defaults to the caller's, which is
110
+ # right for a member's own request thread (:inline). A rooms process has no request tenant,
111
+ # so whatever starts a room there must pass it. The lock key, the room's stream names, and its
112
+ # startup all depend on it, so the whole start runs switched into it.
113
+ def ensure_room(room_class, key = nil, tenant: CableRoom.current_tenant)
114
+ CableRoom.with_tenant(tenant) do
115
+ lock_key = room_class.room_port_key(key)
116
+ # One attempt: a held lock means the room is already running, and retrying (Redlock's
117
+ # default is 3 retries, 200-250 ms apart) would park the caller's ActionCable worker for
118
+ # ~0.7 s on every `create: true` member's ping.
119
+ lock_info = CableRoom.lock_manager.lock(lock_key, room_class::LOCK_DURATION.in_milliseconds, retry_count: 0)
120
+ next false unless lock_info
121
+
122
+ begin
123
+ start_room(
124
+ room_class,
125
+ key,
126
+ lock_info,
127
+ watchdog_interval: room_class::WATCH_DOG_INTERVAL,
128
+ lock_duration: room_class::LOCK_DURATION,
129
+ tenant: tenant,
130
+ )
131
+ rescue => e
132
+ CableRoom.lock_manager.unlock(lock_info)
133
+ raise e
134
+ end
135
+
136
+ true
137
+ end
138
+ end
139
+
140
+ # Start running a room here. The caller (`ensure_room`) has already taken the room's Redlock;
141
+ # the runner renews it from now on and releases it when the room dies.
142
+ def start_room(room_class, key, lock_info, watchdog_interval:, lock_duration:, tenant: CableRoom.current_tenant)
143
+ runner = Runner.new(self, room_class, key, lock_info, watchdog_interval:, lock_duration:, tenant:)
55
144
  runner.start!
56
145
  runner
57
146
  end
@@ -95,6 +184,22 @@ module CableRoom
95
184
  @shutdown
96
185
  end
97
186
 
187
+ # Start hearing provision requests (see Host::Placement). `Host.start!` does this for a :remote
188
+ # rooms process; a Host built with `new` (a spec playing one) calls it itself. `options` go to
189
+ # Placement.new the first time; later calls restart the same Placement.
190
+ def start_placement(**options)
191
+ placement = @monitor.synchronize do
192
+ raise "Cannot start placement after shutdown" if @shutdown
193
+ @placement ||= Placement.new(self, **options)
194
+ end
195
+ placement.start
196
+ end
197
+
198
+ # Stop hearing requests and drop any claim still waiting. Nothing starts a room here after this.
199
+ def stop_placement
200
+ @placement&.stop
201
+ end
202
+
98
203
  # Stop everything: the beat, every room (gracefully, with a bounded wait), then the pool.
99
204
  # Runs once; later calls do nothing.
100
205
  def shutdown!
@@ -103,6 +208,8 @@ module CableRoom
103
208
  @shutdown = true
104
209
  end
105
210
 
211
+ # Stop claiming first: a room started after this point would only be shut down again
212
+ stop_placement
106
213
  scheduler.shutdown
107
214
  shutdown_rooms!
108
215
  worker_pool.executor.shutdown
@@ -110,6 +217,15 @@ module CableRoom
110
217
  worker_pool.executor.kill
111
218
  end
112
219
 
220
+ # Give this Host up without touching anything it holds. For a copy of a Host that a forked
221
+ # child inherited from its parent (see `CableRoom.after_fork!`): its rooms, locks, and threads
222
+ # belong to the parent, so the child must neither close those rooms nor let the `at_exit`
223
+ # hook above do it when the child exits. Marks it shut down so both are no-ops.
224
+ def abandon!
225
+ @shutdown = true
226
+ self
227
+ end
228
+
113
229
  # Ask every room to finish what it has queued and shut down, then give them up to
114
230
  # `wait` seconds to do it
115
231
  def shutdown_rooms!(wait: 15)
@@ -1,4 +1,17 @@
1
1
  module CableRoom
2
+ # The `ports[...]` DSL shared by a Room and by a member's RoomMembership. A port is a named
3
+ # lane between a room and its members; `ports[:x] << msg` sends on it and `ports[:x].stream`
4
+ # listens on it. Which transport that touches depends on which side you're on, and each side
5
+ # implements the three methods below:
6
+ #
7
+ # * Member side (RoomMembership): sending publishes on the room's Bus channel for the port
8
+ # (member→room); listening is an ActionCable stream (room→member).
9
+ # * Room side (Room::Base and Room::HostAdapter): sending is an ActionCable broadcast on the
10
+ # port's stream; listening subscribes on the Bus through the room's Host.
11
+ #
12
+ # So a member's `ports[:custom] << msg` and a room's `ports[:custom].stream` meet on the Bus,
13
+ # and a room's `ports[:custom] << msg` and a member's `ports[:custom].stream` meet on the
14
+ # ActionCable stream named by `room_port_key`. Neither side needs to know that.
2
15
  module Ports
3
16
  extend ActiveSupport::Concern
4
17
 
@@ -6,22 +19,19 @@ module CableRoom
6
19
  @ports_proxy ||= PortsProxy.new(self)
7
20
  end
8
21
 
9
- # The member side, streaming through its ActionCable channel. Rooms override this (and
10
- # close_streamed_ports!) in Room::HostAdapter to listen through their Host::Runner instead.
22
+ # Listen on `port`, handing each decoded message to the block. Ports streamed with
23
+ # `auto_close: true` are stopped by `close_streamed_ports!`.
11
24
  def stream_port(port, auto_close: true, &blk)
12
- @cable_channel.stream_from(room_port_key(port), coder: ActiveSupport::JSON, &blk)
13
- _streamed_ports << port if auto_close
25
+ raise NotImplementedError, "#{self.class} must implement stream_port"
14
26
  end
15
27
 
16
28
  def close_streamed_ports!
17
- _streamed_ports.each do |port|
18
- @cable_channel.stop_stream_from(room_port_key(port))
19
- end
20
- _streamed_ports.clear
29
+ raise NotImplementedError, "#{self.class} must implement close_streamed_ports!"
21
30
  end
22
31
 
32
+ # Send `data` on `port`.
23
33
  def port_transmit(port, data)
24
- ActionCable.server.broadcast(room_port_key(port), data)
34
+ raise NotImplementedError, "#{self.class} must implement port_transmit"
25
35
  end
26
36
 
27
37
  protected
@@ -14,11 +14,17 @@ module CableRoom
14
14
  runner do
15
15
  end
16
16
 
17
+ # Read CABLE_ROOM_HOST now so a bad value fails the boot, not the first room
18
+ initializer "cable_room.room_host" do
19
+ CableRoom.room_host
20
+ end
21
+
17
22
  initializer "cable_room.hook_action_cable_restart" do
18
23
  module ActionCableServerExtensions
19
- # ActionCable restarts when the app reloads; stop every room with it
24
+ # ActionCable restarts when the app reloads; stop every room with it. A process that
25
+ # hosts no rooms (a web process in :remote) has nothing to stop.
20
26
  def restart
21
- CableRoom::Host.instance.stop_all_rooms!
27
+ CableRoom::Host.current&.stop_all_rooms!
22
28
  super
23
29
  end
24
30
  end
@@ -10,26 +10,13 @@ module CableRoom
10
10
  class << self
11
11
  # Start this room in the current process if nobody else is running it. Takes the room's
12
12
  # Redlock first, so exactly one process wins; returns false when the lock is held.
13
- def ensure(key = nil)
14
- lock_key = room_port_key(key)
15
- # One attempt: a held lock means the room is already running, and retrying (Redlock's
16
- # default is 3 retries, 200-250 ms apart) would park the caller's ActionCable worker for
17
- # ~0.7 s on every `create: true` member's ping.
18
- lock_info = CableRoom.lock_manager.lock(lock_key, self::LOCK_DURATION.in_milliseconds, retry_count: 0)
19
- return false unless lock_info
20
-
21
- Host.instance.start_room(
22
- self,
23
- key,
24
- lock_info,
25
- watchdog_interval: self::WATCH_DOG_INTERVAL,
26
- lock_duration: self::LOCK_DURATION,
27
- )
28
-
29
- true
30
- rescue => e
31
- CableRoom.lock_manager.unlock(lock_info) if lock_info
32
- raise e
13
+ # `tenant` defaults to the caller's (see Host#ensure_room).
14
+ #
15
+ # Only a process that hosts rooms can do this: every process in :inline, and only
16
+ # `cable_room server` in :remote. Anywhere else `Host.instance` raises Host::NotHosting
17
+ # rather than letting a room start where it doesn't belong.
18
+ def ensure(key = nil, tenant: CableRoom.current_tenant)
19
+ Host.instance.ensure_room(self, key, tenant: tenant)
33
20
  end
34
21
 
35
22
  # The stream name for a room, or for one of its ports: "RoomClass:key" or
@@ -42,14 +29,23 @@ module CableRoom
42
29
  stream_namer.broadcasting_for(full_key)
43
30
  end
44
31
 
32
+ # The Bus channel members publish on to reach a room's inbound `port`, and the one the
33
+ # room's Host subscribes to for it: `cr:{room_port_key}:in` for the main port and
34
+ # `cr:{room_port_key}:in:{port}` for a custom one (see Bus.inbound_channel).
35
+ def inbound_channel(room_key, port = ROOM_IN_CHANNEL)
36
+ port = port.to_s == ROOM_IN_CHANNEL.to_s ? nil : port.to_s
37
+ Bus.inbound_channel(self, room_key, port)
38
+ end
39
+
40
+ # Send a message to a room from anywhere in the app: the same Bus channel a member uses
45
41
  def send_message(room_key, data, port: ROOM_IN_CHANNEL)
46
- ActionCable.server.broadcast(room_port_key(room_key, port), data)
42
+ CableRoom.bus.publish(inbound_channel(room_key, port), data)
47
43
  end
48
44
 
45
+ # Every instance of this room class running in this process. Empty when the process
46
+ # hosts no rooms (a web process in :remote).
49
47
  def locally_running_instances
50
- Host.instance.runners.select do |runner|
51
- runner.room_class == self
52
- end.map(&:room)
48
+ Room.locally_open_rooms.select { |room| room.class == self }
53
49
  end
54
50
 
55
51
  private
@@ -97,6 +93,13 @@ module CableRoom
97
93
  port_transmit(ROOM_OUT_CHANNEL, data)
98
94
  end
99
95
 
96
+ # The room's sending side of Ports: room→member messages go out as an ActionCable
97
+ # broadcast on the stream named by `room_port_key(port)`. (The listening side is in
98
+ # Room::HostAdapter.) Public because `ports[:x] << msg` reaches it through a PortProxy.
99
+ def port_transmit(port, data)
100
+ ActionCable.server.broadcast(room_port_key(port), data)
101
+ end
102
+
100
103
  protected
101
104
 
102
105
  def room_class
@@ -23,22 +23,27 @@ module CableRoom
23
23
  end
24
24
  end
25
25
 
26
- # The room side of Ports#stream_port: listen on this room's inbound stream for `port`.
27
- # `on_live` runs once the subscription is confirmed.
26
+ # The room's listening side of Ports: subscribe, through the Host, to the Bus channel
27
+ # members publish on for `port` (see Room::Base.inbound_channel). `on_live` runs once the
28
+ # subscription is confirmed.
28
29
  def stream_port(port, auto_close: true, on_live: nil, &blk)
29
- @runner.subscribe(room_port_key(port), coder: ActiveSupport::JSON, on_live: on_live, &blk)
30
+ @runner.subscribe(inbound_channel(port), on_live: on_live, &blk)
30
31
  _streamed_ports << port if auto_close
31
32
  end
32
33
 
33
34
  def close_streamed_ports!
34
35
  _streamed_ports.each do |port|
35
- @runner.unsubscribe(room_port_key(port))
36
+ @runner.unsubscribe(inbound_channel(port))
36
37
  end
37
38
  _streamed_ports.clear
38
39
  end
39
40
 
40
41
  protected
41
42
 
43
+ def inbound_channel(port)
44
+ self.class.inbound_channel(key, port)
45
+ end
46
+
42
47
  def start_periodic_timer(callback, every:)
43
48
  @runner.start_periodic_timer(-> { instance_exec(&callback) }, every: every)
44
49
  end
@@ -20,8 +20,10 @@ module CableRoom
20
20
  autoload :Broadcasting
21
21
  end
22
22
 
23
+ # Every room running in this process. Asking never creates a Host, so a process that hosts
24
+ # no rooms (a web process in :remote) gets an empty list.
23
25
  def self.locally_open_rooms
24
- Host.instance.rooms
26
+ Host.current&.rooms || []
25
27
  end
26
28
  end
27
29
  end