kino 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +21 -0
- data/Cargo.lock +1 -1
- data/README.md +52 -0
- data/doc/architecture.md +10 -0
- data/ext/kino/Cargo.toml +1 -1
- data/ext/kino/src/control.rs +103 -6
- data/ext/kino/src/env_strings.rs +249 -120
- data/ext/kino/src/lib.rs +11 -0
- data/ext/kino/src/pin.rs +21 -9
- data/ext/kino/src/queue.rs +99 -3
- data/ext/kino/src/registry.rs +141 -0
- data/ext/kino/src/request.rs +1 -1
- data/ext/kino/src/server.rs +218 -3
- data/ext/kino/src/test_support.rs +44 -0
- data/lib/kino/cli.rb +12 -4
- data/lib/kino/configuration.rb +16 -2
- data/lib/kino/monitor.rb +52 -0
- data/lib/kino/pool_scaler.rb +103 -0
- data/lib/kino/quarantine_monitor.rb +5 -37
- data/lib/kino/ractor_supervisor.rb +102 -8
- data/lib/kino/server.rb +64 -85
- data/lib/kino/slot_bank.rb +31 -0
- data/lib/kino/templates/kino.rb.tt +15 -0
- data/lib/kino/threaded_pool.rb +186 -0
- data/lib/kino/version.rb +1 -1
- data/lib/kino.rb +4 -0
- data/lib/rackup/handler/kino.rb +4 -4
- data/sig/kino.rbs +2 -0
- metadata +5 -1
data/lib/kino/cli.rb
CHANGED
|
@@ -127,7 +127,7 @@ module Kino
|
|
|
127
127
|
stats = server.stats
|
|
128
128
|
puts dim("- ruby: #{RUBY_DESCRIPTION}")
|
|
129
129
|
puts dim("- env: #{ENV["RAILS_ENV"] || ENV["RACK_ENV"] || "development"}")
|
|
130
|
-
puts dim("- mode: #{server.mode}, #{
|
|
130
|
+
puts dim("- mode: #{server.mode}, #{workers_label(stats)} × #{count(stats[:threads], "thread")}")
|
|
131
131
|
puts dim("- pid: #{Process.pid}")
|
|
132
132
|
puts dim("- listening: #{server.url}")
|
|
133
133
|
puts dim("- control: #{server.control_url}") if server.control_url
|
|
@@ -140,6 +140,14 @@ module Kino
|
|
|
140
140
|
"#{number} #{noun}#{"s" unless number == 1}"
|
|
141
141
|
end
|
|
142
142
|
|
|
143
|
+
# The pool as configured: "8 workers", or "8-32 workers" when it can
|
|
144
|
+
# grow.
|
|
145
|
+
def workers_label(stats)
|
|
146
|
+
return count(stats[:workers], "worker") if stats[:max_workers] == stats[:workers]
|
|
147
|
+
|
|
148
|
+
"#{stats[:workers]}-#{stats[:max_workers]} workers"
|
|
149
|
+
end
|
|
150
|
+
|
|
143
151
|
# Roll credits when the process ends: normal exit or crash (at_exit
|
|
144
152
|
# also runs after an uncaught exception; only a force-exit skips it).
|
|
145
153
|
# @return [void]
|
|
@@ -234,7 +242,7 @@ module Kino
|
|
|
234
242
|
end
|
|
235
243
|
|
|
236
244
|
def write_sample(path)
|
|
237
|
-
require "kino"
|
|
245
|
+
require "kino" # audition:disable runtime-require
|
|
238
246
|
Configuration.write_sample(path)
|
|
239
247
|
puts "Kino: wrote sample config to #{path}"
|
|
240
248
|
0
|
|
@@ -245,8 +253,8 @@ module Kino
|
|
|
245
253
|
|
|
246
254
|
# Resolve the full configuration once: file + CLI flag overrides.
|
|
247
255
|
def resolve_config(options)
|
|
248
|
-
require "kino"
|
|
249
|
-
require "rack"
|
|
256
|
+
require "kino" # audition:disable runtime-require
|
|
257
|
+
require "rack" # audition:disable runtime-require
|
|
250
258
|
|
|
251
259
|
config_file = options[:config_file] || Configuration.default_path
|
|
252
260
|
|
data/lib/kino/configuration.rb
CHANGED
|
@@ -10,6 +10,8 @@ module Kino
|
|
|
10
10
|
bind: "127.0.0.1",
|
|
11
11
|
port: 0,
|
|
12
12
|
workers: nil, # resolved to Kino.available_parallelism in #to_h
|
|
13
|
+
max_workers: nil, # nil = fixed pool; above workers = elastic pool
|
|
14
|
+
scale_down_after: nil, # resolved to 30 seconds in Server
|
|
13
15
|
threads: nil, # resolved per mode in Server: 1 in :ractor, 3 in :threaded
|
|
14
16
|
mode: :auto,
|
|
15
17
|
queue_depth: 1024,
|
|
@@ -44,7 +46,7 @@ module Kino
|
|
|
44
46
|
SETTINGS = DEFAULTS.keys.freeze
|
|
45
47
|
|
|
46
48
|
# Source template for {.sample}.
|
|
47
|
-
SAMPLE_TEMPLATE = File.expand_path("templates/kino.rb.tt", __dir__)
|
|
49
|
+
SAMPLE_TEMPLATE = File.expand_path("templates/kino.rb.tt", __dir__).freeze
|
|
48
50
|
|
|
49
51
|
# Where the `kino` CLI and the Rack handler look for a config file when
|
|
50
52
|
# none is named: the project root first, then the Rails-style config/.
|
|
@@ -146,6 +148,8 @@ module Kino
|
|
|
146
148
|
# bind "0.0.0.0"
|
|
147
149
|
# port 9292
|
|
148
150
|
# workers 8 # ractors (or thread groups in :threaded mode)
|
|
151
|
+
# max_workers 32 # elastic pool ceiling; unset = fixed pool
|
|
152
|
+
# scale_down_after 30 # seconds idle before an extra worker retires
|
|
149
153
|
# threads 3 # threads per worker
|
|
150
154
|
# mode :ractor # :auto | :ractor | :threaded
|
|
151
155
|
# queue_depth 2048
|
|
@@ -172,9 +176,19 @@ module Kino
|
|
|
172
176
|
# Port to listen on; 0 picks an ephemeral port.
|
|
173
177
|
def port(port) = @config.set(:port, Integer(port))
|
|
174
178
|
|
|
175
|
-
# Worker count (ractors in :ractor mode); defaults to CPU cores.
|
|
179
|
+
# Worker count (ractors in :ractor mode); defaults to CPU cores. The
|
|
180
|
+
# pool floor when max_workers is set.
|
|
176
181
|
def workers(count) = @config.set(:workers, Integer(count))
|
|
177
182
|
|
|
183
|
+
# Pool ceiling: under sustained queue pressure the pool grows past
|
|
184
|
+
# `workers`, one worker at a time, up to this many. Unset (the
|
|
185
|
+
# default) keeps the pool fixed at `workers`.
|
|
186
|
+
def max_workers(count) = @config.set(:max_workers, Integer(count))
|
|
187
|
+
|
|
188
|
+
# Seconds a worker above the floor must sit idle before it is
|
|
189
|
+
# retired (default 30). Only meaningful with max_workers.
|
|
190
|
+
def scale_down_after(seconds) = @config.set(:scale_down_after, seconds)
|
|
191
|
+
|
|
178
192
|
# Threads per worker (I/O concurrency inside one ractor); default is
|
|
179
193
|
# mode-dependent: 1 in :ractor mode, 3 in :threaded.
|
|
180
194
|
def threads(count) = @config.set(:threads, Integer(count))
|
data/lib/kino/monitor.rb
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Kino
|
|
4
|
+
# @private
|
|
5
|
+
# A thread on the main ractor that calls `scan` every `tick` seconds
|
|
6
|
+
# until stopped. Main-ractor so it stays responsive when worker ractors
|
|
7
|
+
# are wedged. A scan that raises is logged and skipped, never fatal:
|
|
8
|
+
# monitors keep the server healthy, they must not take it down.
|
|
9
|
+
class Monitor
|
|
10
|
+
def initialize(name:, tick:)
|
|
11
|
+
@name = name
|
|
12
|
+
@tick = tick
|
|
13
|
+
@running = false
|
|
14
|
+
@thread = nil
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
def start
|
|
18
|
+
@running = true
|
|
19
|
+
@thread = Thread.new do
|
|
20
|
+
Thread.current.name = @name
|
|
21
|
+
run
|
|
22
|
+
end
|
|
23
|
+
self
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
def stop
|
|
27
|
+
@running = false
|
|
28
|
+
@thread&.join(@tick * 2)
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
private
|
|
32
|
+
|
|
33
|
+
def run
|
|
34
|
+
tick while @running
|
|
35
|
+
rescue => e
|
|
36
|
+
Log.error("#{@name} crashed: #{e.class}: #{e.message}")
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def tick
|
|
40
|
+
scan
|
|
41
|
+
rescue => e
|
|
42
|
+
Log.error("#{@name} tick error: #{e.class}: #{e.message}")
|
|
43
|
+
ensure
|
|
44
|
+
sleep @tick
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
# One poll; subclasses define it.
|
|
48
|
+
def scan
|
|
49
|
+
raise NotImplementedError, "#{self.class} must define scan"
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
end
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Kino
|
|
4
|
+
# @private
|
|
5
|
+
# Grows and shrinks the worker pool between `floor` and `ceiling`. One
|
|
6
|
+
# thread on the main ractor polls the native queue and per-slot sensors
|
|
7
|
+
# every tick and drives a pool (RactorSupervisor or ThreadedPool) through
|
|
8
|
+
# three methods: `active_count`, `groups` (worker index => slot ids for
|
|
9
|
+
# every worker that may be retired), `grow`, and `retire(index)`.
|
|
10
|
+
#
|
|
11
|
+
# Policy: grow by one worker per tick once requests have been waiting in
|
|
12
|
+
# the queue on two consecutive ticks (a burst that clears within a tick
|
|
13
|
+
# is not pressure); retire one worker per tick, the one idle longest,
|
|
14
|
+
# once it has been idle for `scale_down_after`. "Idle" means no slot of
|
|
15
|
+
# the worker held a request at two consecutive samples and its served
|
|
16
|
+
# count did not move between them, so a worker serving short requests
|
|
17
|
+
# between samples is never mistaken for an idle one.
|
|
18
|
+
class PoolScaler < Monitor
|
|
19
|
+
TICK = 0.1
|
|
20
|
+
# Consecutive ticks with a non-empty queue before the pool grows.
|
|
21
|
+
PRESSURE_TICKS = 2
|
|
22
|
+
|
|
23
|
+
def initialize(server_id:, pool:, floor:, ceiling:, scale_down_after:, tick: TICK)
|
|
24
|
+
super(name: "pool scaler", tick: tick)
|
|
25
|
+
@server_id = server_id
|
|
26
|
+
@pool = pool
|
|
27
|
+
@floor = floor
|
|
28
|
+
@ceiling = ceiling
|
|
29
|
+
@scale_down_after = scale_down_after
|
|
30
|
+
@pressure = 0
|
|
31
|
+
@last_served = {}
|
|
32
|
+
@idle_since = {}
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
# One policy step over one set of observations: the monotonic time,
|
|
36
|
+
# the queue depth, and worker_stats rows ([slot, served, in_flight,
|
|
37
|
+
# busy_ms, quarantined, retired]). Public so the policy is testable
|
|
38
|
+
# without a server.
|
|
39
|
+
def step(now, queued, rows)
|
|
40
|
+
@pressure = queued.positive? ? @pressure + 1 : 0
|
|
41
|
+
if @pressure >= PRESSURE_TICKS
|
|
42
|
+
grow if @pool.active_count < @ceiling
|
|
43
|
+
return
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
observe_idle(now, rows)
|
|
47
|
+
return unless @pool.active_count > @floor
|
|
48
|
+
|
|
49
|
+
index, since = @idle_since.min_by { |_index, at| at }
|
|
50
|
+
return unless index && now - since >= @scale_down_after
|
|
51
|
+
|
|
52
|
+
retire(index)
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
private
|
|
56
|
+
|
|
57
|
+
def scan
|
|
58
|
+
queued, _in_flight = Native.queue_stats(@server_id)
|
|
59
|
+
step(Process.clock_gettime(Process::CLOCK_MONOTONIC), queued, Native.worker_stats(@server_id))
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
def grow
|
|
63
|
+
index = @pool.grow
|
|
64
|
+
return unless index
|
|
65
|
+
|
|
66
|
+
Native.record_scale_up(@server_id)
|
|
67
|
+
Log.info("pool grew to #{@pool.active_count} workers (queue pressure)")
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def retire(index)
|
|
71
|
+
return unless @pool.retire(index)
|
|
72
|
+
|
|
73
|
+
forget(index)
|
|
74
|
+
Native.record_scale_down(@server_id)
|
|
75
|
+
Log.info("pool shrank to #{@pool.active_count} workers (worker-#{index} idle)")
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
# Per worker: idle when no slot holds a request at this sample and the
|
|
79
|
+
# served total did not move since the previous one. First sighting of
|
|
80
|
+
# a worker is never idle (it takes two samples to know).
|
|
81
|
+
def observe_idle(now, rows)
|
|
82
|
+
by_slot = rows.to_h { |row| [row[0], row] }
|
|
83
|
+
groups = @pool.groups
|
|
84
|
+
(@idle_since.keys - groups.keys).each { |index| forget(index) }
|
|
85
|
+
groups.each do |index, slot_ids|
|
|
86
|
+
served = slot_ids.sum { |id| by_slot.dig(id, 1) || 0 }
|
|
87
|
+
busy = slot_ids.any? { |id| (by_slot.dig(id, 2) || 0).positive? }
|
|
88
|
+
idle = !busy && @last_served[index] == served
|
|
89
|
+
@last_served[index] = served
|
|
90
|
+
if idle
|
|
91
|
+
@idle_since[index] ||= now
|
|
92
|
+
else
|
|
93
|
+
@idle_since.delete(index)
|
|
94
|
+
end
|
|
95
|
+
end
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
def forget(index)
|
|
99
|
+
@idle_since.delete(index)
|
|
100
|
+
@last_served.delete(index)
|
|
101
|
+
end
|
|
102
|
+
end
|
|
103
|
+
end
|
|
@@ -3,54 +3,22 @@
|
|
|
3
3
|
module Kino
|
|
4
4
|
# @private
|
|
5
5
|
# Polls per-slot busy_ms and, past the timeout, quarantines a wedged slot
|
|
6
|
-
# and asks the replacer to spawn a fresh worker.
|
|
7
|
-
#
|
|
8
|
-
|
|
9
|
-
class QuarantineMonitor
|
|
6
|
+
# and asks the replacer to spawn a fresh worker. Never interrupts the
|
|
7
|
+
# wedged worker.
|
|
8
|
+
class QuarantineMonitor < Monitor
|
|
10
9
|
def initialize(server_id:, timeout_ms:, max:, replacer:, tick: 0.5)
|
|
10
|
+
super(name: "quarantine monitor", tick: tick)
|
|
11
11
|
@server_id = server_id
|
|
12
12
|
@timeout_ms = timeout_ms
|
|
13
13
|
@max = max
|
|
14
14
|
@replacer = replacer
|
|
15
|
-
@tick = tick
|
|
16
15
|
@outstanding = 0
|
|
17
16
|
@at_cap_logged = false
|
|
18
|
-
@running = false
|
|
19
|
-
@thread = nil
|
|
20
|
-
end
|
|
21
|
-
|
|
22
|
-
def start
|
|
23
|
-
@running = true
|
|
24
|
-
@thread = Thread.new do
|
|
25
|
-
Thread.current.name = "quarantine"
|
|
26
|
-
run
|
|
27
|
-
end
|
|
28
|
-
self
|
|
29
|
-
end
|
|
30
|
-
|
|
31
|
-
def stop
|
|
32
|
-
@running = false
|
|
33
|
-
@thread&.join(@tick * 2)
|
|
34
17
|
end
|
|
35
18
|
|
|
36
19
|
private
|
|
37
20
|
|
|
38
|
-
def
|
|
39
|
-
tick while @running
|
|
40
|
-
rescue => e
|
|
41
|
-
Log.error("quarantine monitor crashed: #{e.class}: #{e.message}")
|
|
42
|
-
end
|
|
43
|
-
|
|
44
|
-
def tick
|
|
45
|
-
scan_slots
|
|
46
|
-
rescue => e
|
|
47
|
-
# A bad tick must never kill the monitor.
|
|
48
|
-
Log.error("quarantine tick error: #{e.class}: #{e.message}")
|
|
49
|
-
ensure
|
|
50
|
-
sleep @tick
|
|
51
|
-
end
|
|
52
|
-
|
|
53
|
-
def scan_slots
|
|
21
|
+
def scan
|
|
54
22
|
Native.worker_stats(@server_id).each do |index, _served, _in_flight, busy_ms, quarantined|
|
|
55
23
|
next if quarantined || busy_ms <= @timeout_ms
|
|
56
24
|
|
|
@@ -5,7 +5,12 @@ module Kino
|
|
|
5
5
|
# Spawns worker ractors and keeps them alive. One supervisor thread per
|
|
6
6
|
# ractor: it blocks in Ractor#value, and a crash (anything that kills the
|
|
7
7
|
# ractor, Exception from app code included) wakes it to 500 the in-flight
|
|
8
|
-
# requests and respawn. Clean exits (queue drained
|
|
8
|
+
# requests and respawn. Clean exits (queue drained at shutdown, or the
|
|
9
|
+
# worker retired by the pool scaler) end supervision.
|
|
10
|
+
#
|
|
11
|
+
# Also the :ractor-mode pool behind PoolScaler: `grow` adds a worker,
|
|
12
|
+
# `retire` sends one home, `groups` lists the ones that may be retired,
|
|
13
|
+
# and `active_count` is what the control plane reports.
|
|
9
14
|
class RactorSupervisor
|
|
10
15
|
def initialize(server_id, app, workers:, threads:, batch: 1, hooks: nil, on_worker_exit: nil)
|
|
11
16
|
@server_id = server_id
|
|
@@ -21,6 +26,12 @@ module Kino
|
|
|
21
26
|
@worker_slots = {}
|
|
22
27
|
@slot_to_worker = {}
|
|
23
28
|
@replaced = {}
|
|
29
|
+
# Worker index => true while its ractor runs, and => true once the
|
|
30
|
+
# scaler asked it to leave. Retired workers hand their slots back to
|
|
31
|
+
# the bank for the next worker to take over.
|
|
32
|
+
@live = {}
|
|
33
|
+
@retiring = {}
|
|
34
|
+
@bank = SlotBank.new(server_id)
|
|
24
35
|
# The first replacement's index; `replace` increments before using it,
|
|
25
36
|
# so this starts one below the first free index (@workers).
|
|
26
37
|
@next_worker_index = @workers - 1
|
|
@@ -28,6 +39,7 @@ module Kino
|
|
|
28
39
|
|
|
29
40
|
def start
|
|
30
41
|
@supervisor_threads = Array.new(@workers) { |index| supervise(index) }
|
|
42
|
+
report_active
|
|
31
43
|
self
|
|
32
44
|
end
|
|
33
45
|
|
|
@@ -52,6 +64,61 @@ module Kino
|
|
|
52
64
|
@lock.synchronize { @supervisor_threads.dup }.each(&:join)
|
|
53
65
|
end
|
|
54
66
|
|
|
67
|
+
# Ractors cannot be force-killed; their clients were already freed by
|
|
68
|
+
# abort_all_inflight. A stuck ractor leaks until process exit.
|
|
69
|
+
def kill_stragglers
|
|
70
|
+
Log.error("shutdown deadline passed with stuck ractor workers") unless done?
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
# Workers that are staying: the live ones minus those the quarantine
|
|
74
|
+
# monitor abandoned as wedged (their replacements count instead) and
|
|
75
|
+
# minus those already told to leave. A retiring worker leaves the
|
|
76
|
+
# count at once, not when its ractor finally exits, so the scaler
|
|
77
|
+
# never sees a stale surplus and retires past its floor.
|
|
78
|
+
def active_count
|
|
79
|
+
@lock.synchronize do
|
|
80
|
+
@live.count { |index, _| !@replaced.key?(index) && !@retiring.key?(index) }
|
|
81
|
+
end
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
# Worker index => slot ids for every worker the scaler may retire:
|
|
85
|
+
# live, not already leaving, not quarantined, and past its spawn (a
|
|
86
|
+
# worker marked live whose supervisor thread has not assigned slots
|
|
87
|
+
# yet is not listed until it has).
|
|
88
|
+
def groups
|
|
89
|
+
@lock.synchronize do
|
|
90
|
+
@live.keys
|
|
91
|
+
.reject { |index| @retiring.key?(index) || @replaced.key?(index) || !@worker_slots.key?(index) }
|
|
92
|
+
.to_h { |index| [index, @worker_slots[index].dup] }
|
|
93
|
+
end
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
# Add one supervised worker; returns its index.
|
|
97
|
+
def grow
|
|
98
|
+
new_index = @lock.synchronize { @next_worker_index += 1 }
|
|
99
|
+
thread = supervise(new_index)
|
|
100
|
+
@lock.synchronize { @supervisor_threads << thread }
|
|
101
|
+
report_active
|
|
102
|
+
new_index
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
# Send a worker home. Its slots stop receiving work now; the worker
|
|
106
|
+
# finishes what it holds, leaves at its next idle tick, and its slots
|
|
107
|
+
# come back to the free list once the ractor has exited. Returns
|
|
108
|
+
# false when there is no such live worker to retire.
|
|
109
|
+
def retire(worker_index)
|
|
110
|
+
slot_ids = @lock.synchronize do
|
|
111
|
+
next nil unless @live.key?(worker_index) && !@retiring.key?(worker_index)
|
|
112
|
+
|
|
113
|
+
@retiring[worker_index] = true
|
|
114
|
+
@worker_slots[worker_index]
|
|
115
|
+
end
|
|
116
|
+
return false unless slot_ids
|
|
117
|
+
|
|
118
|
+
slot_ids.each { |id| Native.retire_slot(@server_id, id) }
|
|
119
|
+
true
|
|
120
|
+
end
|
|
121
|
+
|
|
55
122
|
# Replace the ractor owning slot `worker_id`: spawn a fresh supervised
|
|
56
123
|
# ractor, then quarantine the old ractor's slots. The old supervisor
|
|
57
124
|
# thread stays blocked in ractor.value on the wedged ractor (it and the
|
|
@@ -90,6 +157,9 @@ module Kino
|
|
|
90
157
|
private
|
|
91
158
|
|
|
92
159
|
def supervise(index)
|
|
160
|
+
# Live from the moment it is asked for, not from when its thread gets
|
|
161
|
+
# around to spawning: `grow` reports the count right after this.
|
|
162
|
+
@lock.synchronize { @live[index] = true }
|
|
93
163
|
Thread.new do
|
|
94
164
|
Thread.current.name = "supervisor-#{index}"
|
|
95
165
|
crashes = 0
|
|
@@ -97,8 +167,9 @@ module Kino
|
|
|
97
167
|
ractor, worker_ids = spawn_worker(index)
|
|
98
168
|
begin
|
|
99
169
|
ractor.value # blocks until the ractor terminates
|
|
100
|
-
HookFire.fire(@on_worker_exit, "on_worker_exit", index, nil) # clean exit:
|
|
101
|
-
|
|
170
|
+
HookFire.fire(@on_worker_exit, "on_worker_exit", index, nil) # clean exit: drained or retired
|
|
171
|
+
exited(index, worker_ids)
|
|
172
|
+
break
|
|
102
173
|
rescue Ractor::Error => e
|
|
103
174
|
# The ractor died mid-flight. Anything it was serving will never
|
|
104
175
|
# be answered by Ruby: 500 those clients NOW (not when GC gets
|
|
@@ -106,11 +177,17 @@ module Kino
|
|
|
106
177
|
worker_ids.each { |id| Native.abort_inflight(@server_id, id) }
|
|
107
178
|
cause = (e.respond_to?(:cause) && e.cause) ? e.cause : e
|
|
108
179
|
HookFire.fire(@on_worker_exit, "on_worker_exit", index, cause)
|
|
109
|
-
|
|
180
|
+
if draining?
|
|
181
|
+
exited(index, nil)
|
|
182
|
+
break
|
|
183
|
+
end
|
|
110
184
|
|
|
111
185
|
crashes += 1
|
|
112
186
|
Native.record_respawn(@server_id)
|
|
113
187
|
Log.error("worker-#{index} crashed (#{cause.class}: #{cause.message}); respawning")
|
|
188
|
+
# A crashed worker that was on its way out respawns on fresh
|
|
189
|
+
# slots like any other; the scaler retires it again when idle.
|
|
190
|
+
@lock.synchronize { @retiring.delete(index) }
|
|
114
191
|
# Policy (crash recovery): unlimited respawn
|
|
115
192
|
# keeps the server up under rare crashes but turns a
|
|
116
193
|
# crash-on-every-request bug into a busy loop. A circuit breaker
|
|
@@ -121,11 +198,10 @@ module Kino
|
|
|
121
198
|
end
|
|
122
199
|
end
|
|
123
200
|
|
|
124
|
-
# Fresh ractor
|
|
125
|
-
#
|
|
126
|
-
# old slot.
|
|
201
|
+
# Fresh ractor on slots from the bank: fresh ones, or ones a retired
|
|
202
|
+
# worker handed back.
|
|
127
203
|
def spawn_worker(worker_index)
|
|
128
|
-
worker_ids = Array.new(@threads) {
|
|
204
|
+
worker_ids = Array.new(@threads) { @bank.claim }
|
|
129
205
|
@lock.synchronize do
|
|
130
206
|
@worker_slots[worker_index] = worker_ids
|
|
131
207
|
worker_ids.each { |id| @slot_to_worker[id] = worker_index }
|
|
@@ -147,6 +223,24 @@ module Kino
|
|
|
147
223
|
[ractor, worker_ids]
|
|
148
224
|
end
|
|
149
225
|
|
|
226
|
+
# Bookkeeping for a supervisor thread that is done: the worker is no
|
|
227
|
+
# longer live, a retired worker's slots go back to the bank, and the
|
|
228
|
+
# thread leaves the join set so a long-lived elastic pool does not
|
|
229
|
+
# accumulate dead threads.
|
|
230
|
+
def exited(index, worker_ids)
|
|
231
|
+
retired = @lock.synchronize do
|
|
232
|
+
@live.delete(index)
|
|
233
|
+
@supervisor_threads.delete(Thread.current)
|
|
234
|
+
@retiring.delete(index)
|
|
235
|
+
end
|
|
236
|
+
@bank.release(worker_ids) if retired && worker_ids
|
|
237
|
+
report_active
|
|
238
|
+
end
|
|
239
|
+
|
|
240
|
+
def report_active
|
|
241
|
+
Native.set_active_workers(@server_id, active_count)
|
|
242
|
+
end
|
|
243
|
+
|
|
150
244
|
def draining?
|
|
151
245
|
@lock.synchronize { @draining }
|
|
152
246
|
end
|