kino 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/lib/kino/cli.rb CHANGED
@@ -127,7 +127,7 @@ module Kino
127
127
  stats = server.stats
128
128
  puts dim("- ruby: #{RUBY_DESCRIPTION}")
129
129
  puts dim("- env: #{ENV["RAILS_ENV"] || ENV["RACK_ENV"] || "development"}")
130
- puts dim("- mode: #{server.mode}, #{count(stats[:workers], "worker")} × #{count(stats[:threads], "thread")}")
130
+ puts dim("- mode: #{server.mode}, #{workers_label(stats)} × #{count(stats[:threads], "thread")}")
131
131
  puts dim("- pid: #{Process.pid}")
132
132
  puts dim("- listening: #{server.url}")
133
133
  puts dim("- control: #{server.control_url}") if server.control_url
@@ -140,6 +140,14 @@ module Kino
140
140
  "#{number} #{noun}#{"s" unless number == 1}"
141
141
  end
142
142
 
143
+ # The pool as configured: "8 workers", or "8-32 workers" when it can
144
+ # grow.
145
+ def workers_label(stats)
146
+ return count(stats[:workers], "worker") if stats[:max_workers] == stats[:workers]
147
+
148
+ "#{stats[:workers]}-#{stats[:max_workers]} workers"
149
+ end
150
+
143
151
  # Roll credits when the process ends: normal exit or crash (at_exit
144
152
  # also runs after an uncaught exception; only a force-exit skips it).
145
153
  # @return [void]
@@ -234,7 +242,7 @@ module Kino
234
242
  end
235
243
 
236
244
  def write_sample(path)
237
- require "kino"
245
+ require "kino" # audition:disable runtime-require
238
246
  Configuration.write_sample(path)
239
247
  puts "Kino: wrote sample config to #{path}"
240
248
  0
@@ -245,8 +253,8 @@ module Kino
245
253
 
246
254
  # Resolve the full configuration once: file + CLI flag overrides.
247
255
  def resolve_config(options)
248
- require "kino"
249
- require "rack"
256
+ require "kino" # audition:disable runtime-require
257
+ require "rack" # audition:disable runtime-require
250
258
 
251
259
  config_file = options[:config_file] || Configuration.default_path
252
260
 
@@ -10,6 +10,8 @@ module Kino
10
10
  bind: "127.0.0.1",
11
11
  port: 0,
12
12
  workers: nil, # resolved to Kino.available_parallelism in #to_h
13
+ max_workers: nil, # nil = fixed pool; above workers = elastic pool
14
+ scale_down_after: nil, # resolved to 30 seconds in Server
13
15
  threads: nil, # resolved per mode in Server: 1 in :ractor, 3 in :threaded
14
16
  mode: :auto,
15
17
  queue_depth: 1024,
@@ -44,7 +46,7 @@ module Kino
44
46
  SETTINGS = DEFAULTS.keys.freeze
45
47
 
46
48
  # Source template for {.sample}.
47
- SAMPLE_TEMPLATE = File.expand_path("templates/kino.rb.tt", __dir__)
49
+ SAMPLE_TEMPLATE = File.expand_path("templates/kino.rb.tt", __dir__).freeze
48
50
 
49
51
  # Where the `kino` CLI and the Rack handler look for a config file when
50
52
  # none is named: the project root first, then the Rails-style config/.
@@ -146,6 +148,8 @@ module Kino
146
148
  # bind "0.0.0.0"
147
149
  # port 9292
148
150
  # workers 8 # ractors (or thread groups in :threaded mode)
151
+ # max_workers 32 # elastic pool ceiling; unset = fixed pool
152
+ # scale_down_after 30 # seconds idle before an extra worker retires
149
153
  # threads 3 # threads per worker
150
154
  # mode :ractor # :auto | :ractor | :threaded
151
155
  # queue_depth 2048
@@ -172,9 +176,19 @@ module Kino
172
176
  # Port to listen on; 0 picks an ephemeral port.
173
177
  def port(port) = @config.set(:port, Integer(port))
174
178
 
175
- # Worker count (ractors in :ractor mode); defaults to CPU cores.
179
+ # Worker count (ractors in :ractor mode); defaults to CPU cores. The
180
+ # pool floor when max_workers is set.
176
181
  def workers(count) = @config.set(:workers, Integer(count))
177
182
 
183
+ # Pool ceiling: under sustained queue pressure the pool grows past
184
+ # `workers`, one worker at a time, up to this many. Unset (the
185
+ # default) keeps the pool fixed at `workers`.
186
+ def max_workers(count) = @config.set(:max_workers, Integer(count))
187
+
188
+ # Seconds a worker above the floor must sit idle before it is
189
+ # retired (default 30). Only meaningful with max_workers.
190
+ def scale_down_after(seconds) = @config.set(:scale_down_after, seconds)
191
+
178
192
  # Threads per worker (I/O concurrency inside one ractor); default is
179
193
  # mode-dependent: 1 in :ractor mode, 3 in :threaded.
180
194
  def threads(count) = @config.set(:threads, Integer(count))
@@ -0,0 +1,52 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Kino
4
+ # @private
5
+ # A thread on the main ractor that calls `scan` every `tick` seconds
6
+ # until stopped. Main-ractor so it stays responsive when worker ractors
7
+ # are wedged. A scan that raises is logged and skipped, never fatal:
8
+ # monitors keep the server healthy, they must not take it down.
9
+ class Monitor
10
+ def initialize(name:, tick:)
11
+ @name = name
12
+ @tick = tick
13
+ @running = false
14
+ @thread = nil
15
+ end
16
+
17
+ def start
18
+ @running = true
19
+ @thread = Thread.new do
20
+ Thread.current.name = @name
21
+ run
22
+ end
23
+ self
24
+ end
25
+
26
+ def stop
27
+ @running = false
28
+ @thread&.join(@tick * 2)
29
+ end
30
+
31
+ private
32
+
33
+ def run
34
+ tick while @running
35
+ rescue => e
36
+ Log.error("#{@name} crashed: #{e.class}: #{e.message}")
37
+ end
38
+
39
+ def tick
40
+ scan
41
+ rescue => e
42
+ Log.error("#{@name} tick error: #{e.class}: #{e.message}")
43
+ ensure
44
+ sleep @tick
45
+ end
46
+
47
+ # One poll; subclasses define it.
48
+ def scan
49
+ raise NotImplementedError, "#{self.class} must define scan"
50
+ end
51
+ end
52
+ end
@@ -0,0 +1,103 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Kino
4
+ # @private
5
+ # Grows and shrinks the worker pool between `floor` and `ceiling`. One
6
+ # thread on the main ractor polls the native queue and per-slot sensors
7
+ # every tick and drives a pool (RactorSupervisor or ThreadedPool) through
8
+ # three methods: `active_count`, `groups` (worker index => slot ids for
9
+ # every worker that may be retired), `grow`, and `retire(index)`.
10
+ #
11
+ # Policy: grow by one worker per tick once requests have been waiting in
12
+ # the queue on two consecutive ticks (a burst that clears within a tick
13
+ # is not pressure); retire one worker per tick, the one idle longest,
14
+ # once it has been idle for `scale_down_after`. "Idle" means no slot of
15
+ # the worker held a request at two consecutive samples and its served
16
+ # count did not move between them, so a worker serving short requests
17
+ # between samples is never mistaken for an idle one.
18
+ class PoolScaler < Monitor
19
+ TICK = 0.1
20
+ # Consecutive ticks with a non-empty queue before the pool grows.
21
+ PRESSURE_TICKS = 2
22
+
23
+ def initialize(server_id:, pool:, floor:, ceiling:, scale_down_after:, tick: TICK)
24
+ super(name: "pool scaler", tick: tick)
25
+ @server_id = server_id
26
+ @pool = pool
27
+ @floor = floor
28
+ @ceiling = ceiling
29
+ @scale_down_after = scale_down_after
30
+ @pressure = 0
31
+ @last_served = {}
32
+ @idle_since = {}
33
+ end
34
+
35
+ # One policy step over one set of observations: the monotonic time,
36
+ # the queue depth, and worker_stats rows ([slot, served, in_flight,
37
+ # busy_ms, quarantined, retired]). Public so the policy is testable
38
+ # without a server.
39
+ def step(now, queued, rows)
40
+ @pressure = queued.positive? ? @pressure + 1 : 0
41
+ if @pressure >= PRESSURE_TICKS
42
+ grow if @pool.active_count < @ceiling
43
+ return
44
+ end
45
+
46
+ observe_idle(now, rows)
47
+ return unless @pool.active_count > @floor
48
+
49
+ index, since = @idle_since.min_by { |_index, at| at }
50
+ return unless index && now - since >= @scale_down_after
51
+
52
+ retire(index)
53
+ end
54
+
55
+ private
56
+
57
+ def scan
58
+ queued, _in_flight = Native.queue_stats(@server_id)
59
+ step(Process.clock_gettime(Process::CLOCK_MONOTONIC), queued, Native.worker_stats(@server_id))
60
+ end
61
+
62
+ def grow
63
+ index = @pool.grow
64
+ return unless index
65
+
66
+ Native.record_scale_up(@server_id)
67
+ Log.info("pool grew to #{@pool.active_count} workers (queue pressure)")
68
+ end
69
+
70
+ def retire(index)
71
+ return unless @pool.retire(index)
72
+
73
+ forget(index)
74
+ Native.record_scale_down(@server_id)
75
+ Log.info("pool shrank to #{@pool.active_count} workers (worker-#{index} idle)")
76
+ end
77
+
78
+ # Per worker: idle when no slot holds a request at this sample and the
79
+ # served total did not move since the previous one. First sighting of
80
+ # a worker is never idle (it takes two samples to know).
81
+ def observe_idle(now, rows)
82
+ by_slot = rows.to_h { |row| [row[0], row] }
83
+ groups = @pool.groups
84
+ (@idle_since.keys - groups.keys).each { |index| forget(index) }
85
+ groups.each do |index, slot_ids|
86
+ served = slot_ids.sum { |id| by_slot.dig(id, 1) || 0 }
87
+ busy = slot_ids.any? { |id| (by_slot.dig(id, 2) || 0).positive? }
88
+ idle = !busy && @last_served[index] == served
89
+ @last_served[index] = served
90
+ if idle
91
+ @idle_since[index] ||= now
92
+ else
93
+ @idle_since.delete(index)
94
+ end
95
+ end
96
+ end
97
+
98
+ def forget(index)
99
+ @idle_since.delete(index)
100
+ @last_served.delete(index)
101
+ end
102
+ end
103
+ end
@@ -3,54 +3,22 @@
3
3
  module Kino
4
4
  # @private
5
5
  # Polls per-slot busy_ms and, past the timeout, quarantines a wedged slot
6
- # and asks the replacer to spawn a fresh worker. Runs one thread on the
7
- # main ractor (uncontended by wedged worker ractors, so it stays
8
- # responsive in :ractor mode). Never interrupts the wedged worker.
9
- class QuarantineMonitor
6
+ # and asks the replacer to spawn a fresh worker. Never interrupts the
7
+ # wedged worker.
8
+ class QuarantineMonitor < Monitor
10
9
  def initialize(server_id:, timeout_ms:, max:, replacer:, tick: 0.5)
10
+ super(name: "quarantine monitor", tick: tick)
11
11
  @server_id = server_id
12
12
  @timeout_ms = timeout_ms
13
13
  @max = max
14
14
  @replacer = replacer
15
- @tick = tick
16
15
  @outstanding = 0
17
16
  @at_cap_logged = false
18
- @running = false
19
- @thread = nil
20
- end
21
-
22
- def start
23
- @running = true
24
- @thread = Thread.new do
25
- Thread.current.name = "quarantine"
26
- run
27
- end
28
- self
29
- end
30
-
31
- def stop
32
- @running = false
33
- @thread&.join(@tick * 2)
34
17
  end
35
18
 
36
19
  private
37
20
 
38
- def run
39
- tick while @running
40
- rescue => e
41
- Log.error("quarantine monitor crashed: #{e.class}: #{e.message}")
42
- end
43
-
44
- def tick
45
- scan_slots
46
- rescue => e
47
- # A bad tick must never kill the monitor.
48
- Log.error("quarantine tick error: #{e.class}: #{e.message}")
49
- ensure
50
- sleep @tick
51
- end
52
-
53
- def scan_slots
21
+ def scan
54
22
  Native.worker_stats(@server_id).each do |index, _served, _in_flight, busy_ms, quarantined|
55
23
  next if quarantined || busy_ms <= @timeout_ms
56
24
 
@@ -5,7 +5,12 @@ module Kino
5
5
  # Spawns worker ractors and keeps them alive. One supervisor thread per
6
6
  # ractor: it blocks in Ractor#value, and a crash (anything that kills the
7
7
  # ractor, Exception from app code included) wakes it to 500 the in-flight
8
- # requests and respawn. Clean exits (queue drained) end supervision.
8
+ # requests and respawn. Clean exits (queue drained at shutdown, or the
9
+ # worker retired by the pool scaler) end supervision.
10
+ #
11
+ # Also the :ractor-mode pool behind PoolScaler: `grow` adds a worker,
12
+ # `retire` sends one home, `groups` lists the ones that may be retired,
13
+ # and `active_count` is what the control plane reports.
9
14
  class RactorSupervisor
10
15
  def initialize(server_id, app, workers:, threads:, batch: 1, hooks: nil, on_worker_exit: nil)
11
16
  @server_id = server_id
@@ -21,6 +26,12 @@ module Kino
21
26
  @worker_slots = {}
22
27
  @slot_to_worker = {}
23
28
  @replaced = {}
29
+ # Worker index => true while its ractor runs, and => true once the
30
+ # scaler asked it to leave. Retired workers hand their slots back to
31
+ # the bank for the next worker to take over.
32
+ @live = {}
33
+ @retiring = {}
34
+ @bank = SlotBank.new(server_id)
24
35
  # The first replacement's index; `replace` increments before using it,
25
36
  # so this starts one below the first free index (@workers).
26
37
  @next_worker_index = @workers - 1
@@ -28,6 +39,7 @@ module Kino
28
39
 
29
40
  def start
30
41
  @supervisor_threads = Array.new(@workers) { |index| supervise(index) }
42
+ report_active
31
43
  self
32
44
  end
33
45
 
@@ -52,6 +64,61 @@ module Kino
52
64
  @lock.synchronize { @supervisor_threads.dup }.each(&:join)
53
65
  end
54
66
 
67
+ # Ractors cannot be force-killed; their clients were already freed by
68
+ # abort_all_inflight. A stuck ractor leaks until process exit.
69
+ def kill_stragglers
70
+ Log.error("shutdown deadline passed with stuck ractor workers") unless done?
71
+ end
72
+
73
+ # Workers that are staying: the live ones minus those the quarantine
74
+ # monitor abandoned as wedged (their replacements count instead) and
75
+ # minus those already told to leave. A retiring worker leaves the
76
+ # count at once, not when its ractor finally exits, so the scaler
77
+ # never sees a stale surplus and retires past its floor.
78
+ def active_count
79
+ @lock.synchronize do
80
+ @live.count { |index, _| !@replaced.key?(index) && !@retiring.key?(index) }
81
+ end
82
+ end
83
+
84
+ # Worker index => slot ids for every worker the scaler may retire:
85
+ # live, not already leaving, not quarantined, and past its spawn (a
86
+ # worker marked live whose supervisor thread has not assigned slots
87
+ # yet is not listed until it has).
88
+ def groups
89
+ @lock.synchronize do
90
+ @live.keys
91
+ .reject { |index| @retiring.key?(index) || @replaced.key?(index) || !@worker_slots.key?(index) }
92
+ .to_h { |index| [index, @worker_slots[index].dup] }
93
+ end
94
+ end
95
+
96
+ # Add one supervised worker; returns its index.
97
+ def grow
98
+ new_index = @lock.synchronize { @next_worker_index += 1 }
99
+ thread = supervise(new_index)
100
+ @lock.synchronize { @supervisor_threads << thread }
101
+ report_active
102
+ new_index
103
+ end
104
+
105
+ # Send a worker home. Its slots stop receiving work now; the worker
106
+ # finishes what it holds, leaves at its next idle tick, and its slots
107
+ # come back to the free list once the ractor has exited. Returns
108
+ # false when there is no such live worker to retire.
109
+ def retire(worker_index)
110
+ slot_ids = @lock.synchronize do
111
+ next nil unless @live.key?(worker_index) && !@retiring.key?(worker_index)
112
+
113
+ @retiring[worker_index] = true
114
+ @worker_slots[worker_index]
115
+ end
116
+ return false unless slot_ids
117
+
118
+ slot_ids.each { |id| Native.retire_slot(@server_id, id) }
119
+ true
120
+ end
121
+
55
122
  # Replace the ractor owning slot `worker_id`: spawn a fresh supervised
56
123
  # ractor, then quarantine the old ractor's slots. The old supervisor
57
124
  # thread stays blocked in ractor.value on the wedged ractor (it and the
@@ -90,6 +157,9 @@ module Kino
90
157
  private
91
158
 
92
159
  def supervise(index)
160
+ # Live from the moment it is asked for, not from when its thread gets
161
+ # around to spawning: `grow` reports the count right after this.
162
+ @lock.synchronize { @live[index] = true }
93
163
  Thread.new do
94
164
  Thread.current.name = "supervisor-#{index}"
95
165
  crashes = 0
@@ -97,8 +167,9 @@ module Kino
97
167
  ractor, worker_ids = spawn_worker(index)
98
168
  begin
99
169
  ractor.value # blocks until the ractor terminates
100
- HookFire.fire(@on_worker_exit, "on_worker_exit", index, nil) # clean exit: queue drained
101
- break # clean exit: queue closed, workers drained
170
+ HookFire.fire(@on_worker_exit, "on_worker_exit", index, nil) # clean exit: drained or retired
171
+ exited(index, worker_ids)
172
+ break
102
173
  rescue Ractor::Error => e
103
174
  # The ractor died mid-flight. Anything it was serving will never
104
175
  # be answered by Ruby: 500 those clients NOW (not when GC gets
@@ -106,11 +177,17 @@ module Kino
106
177
  worker_ids.each { |id| Native.abort_inflight(@server_id, id) }
107
178
  cause = (e.respond_to?(:cause) && e.cause) ? e.cause : e
108
179
  HookFire.fire(@on_worker_exit, "on_worker_exit", index, cause)
109
- break if draining?
180
+ if draining?
181
+ exited(index, nil)
182
+ break
183
+ end
110
184
 
111
185
  crashes += 1
112
186
  Native.record_respawn(@server_id)
113
187
  Log.error("worker-#{index} crashed (#{cause.class}: #{cause.message}); respawning")
188
+ # A crashed worker that was on its way out respawns on fresh
189
+ # slots like any other; the scaler retires it again when idle.
190
+ @lock.synchronize { @retiring.delete(index) }
114
191
  # Policy (crash recovery): unlimited respawn
115
192
  # keeps the server up under rare crashes but turns a
116
193
  # crash-on-every-request bug into a busy loop. A circuit breaker
@@ -121,11 +198,10 @@ module Kino
121
198
  end
122
199
  end
123
200
 
124
- # Fresh ractor + fresh native slots. Slots are never reused across
125
- # respawns: stale interrupt kicks and dead weak refs go down with the
126
- # old slot.
201
+ # Fresh ractor on slots from the bank: fresh ones, or ones a retired
202
+ # worker handed back.
127
203
  def spawn_worker(worker_index)
128
- worker_ids = Array.new(@threads) { Native.register_worker(@server_id) }
204
+ worker_ids = Array.new(@threads) { @bank.claim }
129
205
  @lock.synchronize do
130
206
  @worker_slots[worker_index] = worker_ids
131
207
  worker_ids.each { |id| @slot_to_worker[id] = worker_index }
@@ -147,6 +223,24 @@ module Kino
147
223
  [ractor, worker_ids]
148
224
  end
149
225
 
226
+ # Bookkeeping for a supervisor thread that is done: the worker is no
227
+ # longer live, a retired worker's slots go back to the bank, and the
228
+ # thread leaves the join set so a long-lived elastic pool does not
229
+ # accumulate dead threads.
230
+ def exited(index, worker_ids)
231
+ retired = @lock.synchronize do
232
+ @live.delete(index)
233
+ @supervisor_threads.delete(Thread.current)
234
+ @retiring.delete(index)
235
+ end
236
+ @bank.release(worker_ids) if retired && worker_ids
237
+ report_active
238
+ end
239
+
240
+ def report_active
241
+ Native.set_active_workers(@server_id, active_count)
242
+ end
243
+
150
244
  def draining?
151
245
  @lock.synchronize { @draining }
152
246
  end