kino 0.2.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/lib/kino/server.rb CHANGED
@@ -12,6 +12,10 @@ module Kino
12
12
  # port when configured with port 0)
13
13
  attr_reader :port
14
14
 
15
+ # @return [Integer, nil] the control plane's TCP port (nil until #start,
16
+ # when the control plane is off, or for a unix-socket bind)
17
+ attr_reader :control_port
18
+
15
19
  # @return [Symbol] the resolved dispatch mode, :ractor or :threaded
16
20
  attr_reader :mode
17
21
 
@@ -23,6 +27,29 @@ module Kino
23
27
  !@tls.nil?
24
28
  end
25
29
 
30
+ # @return [Boolean] whether the bind is a unix domain socket
31
+ # ("unix:///path/to.sock")
32
+ def unix?
33
+ @bind.start_with?("unix://")
34
+ end
35
+
36
+ # Where the server listens, once started: `http://host:port`
37
+ # (`https` under TLS), or the `unix://` socket path.
38
+ # @return [String]
39
+ def url
40
+ unix? ? @bind : "http#{"s" if tls?}://#{@bind}:#{@port}"
41
+ end
42
+
43
+ # Where the control plane listens, once started, or nil when it is
44
+ # off: `http://host:port`, or its `unix://` socket path.
45
+ # @return [String, nil]
46
+ def control_url
47
+ return nil unless @control_bind
48
+ return @control_bind if @control_bind.start_with?("unix://")
49
+
50
+ "http://#{@control_bind.rpartition(":").first}:#{@control_port}"
51
+ end
52
+
26
53
  # Settings precedence: explicit kwargs > config_file DSL > defaults.
27
54
  #
28
55
  # @param app [#call] a Rack 3 application
@@ -41,8 +68,22 @@ module Kino
41
68
  @bind = settings[:bind]
42
69
  @requested_port = settings[:port]
43
70
  @workers = Integer(settings[:workers])
44
- @on_error = validate_on_error(settings[:on_error])
71
+ @on_error = validate_hook(settings[:on_error], :on_error)
72
+ @after_worker_boot = validate_hook(settings[:after_worker_boot], :after_worker_boot)
73
+ @after_request_complete = validate_hook(settings[:after_request_complete], :after_request_complete)
74
+ @after_boot = validate_hook(settings[:after_boot], :after_boot)
75
+ @on_worker_exit = validate_hook(settings[:on_worker_exit], :on_worker_exit)
45
76
  @mode = resolve_mode(settings[:mode])
77
+ @worker_hooks = WorkerHooks.new(
78
+ on_error: @on_error,
79
+ after_worker_boot: @after_worker_boot,
80
+ after_request_complete: @after_request_complete,
81
+ # The access log's GC and allocation figures come from the VM's
82
+ # process-wide counters, so they are measured only where one
83
+ # request at a time can own them: the GVL serializes :threaded
84
+ # mode, and a single ractor has nothing to race.
85
+ access_timing: !!settings[:log_requests] && (@mode == :threaded || @workers == 1)
86
+ )
46
87
  # Default threads per mode: 1 in :ractor (threads inside a ractor
47
88
  # share its lock; a measured +17% on fast handlers; raise `workers`
48
89
  # for I/O concurrency instead), 3 in :threaded (threads ARE the
@@ -59,9 +100,29 @@ module Kino
59
100
  @shutdown_timeout = settings[:shutdown_timeout]
60
101
  @tokio_threads = settings[:tokio_threads]
61
102
  @tls = validate_tls(settings[:tls])
103
+ if @tls && unix?
104
+ raise ArgumentError, "TLS is not supported on a unix socket bind; terminate TLS at the proxy in front"
105
+ end
62
106
  @pidfile = settings[:pidfile]
107
+ @control_bind = settings[:control_bind]&.to_s
108
+ @control_token = settings[:control_token]&.to_s
109
+ # An empty token (e.g. control_token ENV["KINO_CONTROL_TOKEN"] with the
110
+ # var unset) must not half-disable auth: treat it as auth off, not as
111
+ # "require a zero-length Bearer token".
112
+ @control_token = nil if @control_token && @control_token.empty?
113
+ @quarantine_timeout_ms = settings[:quarantine_timeout] ? (Float(settings[:quarantine_timeout]) * 1000).round : nil
114
+ @quarantine_max =
115
+ if settings[:quarantine_max]
116
+ Integer(settings[:quarantine_max])
117
+ elsif @mode == :ractor
118
+ @workers
119
+ else
120
+ @workers * @threads
121
+ end
63
122
  @worker_threads = []
123
+ @worker_threads_lock = Mutex.new
64
124
  @supervisor = nil
125
+ @quarantine_monitor = nil
65
126
  @started = false
66
127
  end
67
128
 
@@ -78,7 +139,7 @@ module Kino
78
139
  write_pidfile if @pidfile
79
140
  booted = false
80
141
  begin
81
- @id, @port = Native.server_start(
142
+ @id, @port, @control_port = Native.server_start(
82
143
  bind: @bind, port: @requested_port,
83
144
  queue_depth: @queue_depth, queue_timeout_ms: @queue_timeout_ms,
84
145
  request_timeout_ms: @request_timeout_ms,
@@ -86,7 +147,9 @@ module Kino
86
147
  max_body_size: @max_body_size,
87
148
  tokio_threads: @tokio_threads,
88
149
  tls_cert: @tls&.fetch(:cert), tls_key: @tls&.fetch(:key),
89
- lanes: @lanes, log_requests: @log_requests
150
+ lanes: @lanes, log_requests: @log_requests,
151
+ mode: @mode.to_s, workers: @workers, threads: @threads, batch: @batch,
152
+ control_bind: @control_bind, control_token: @control_token
90
153
  )
91
154
  booted = true
92
155
  ensure
@@ -97,13 +160,13 @@ module Kino
97
160
  @pin_keeper = Native.pin_keeper(@id)
98
161
  if @mode == :ractor
99
162
  @supervisor = RactorSupervisor.new(@id, @app, workers: @workers, threads: @threads,
100
- batch: @batch, on_error: @on_error).start
163
+ batch: @batch, hooks: @worker_hooks, on_worker_exit: @on_worker_exit).start
101
164
  else
102
- @worker_threads = (@workers * @threads).times.map do
103
- worker_id = Native.register_worker(@id)
104
- Thread.new { Worker.run(@id, worker_id, @app, @batch, @on_error) }
105
- end
165
+ @worker_threads = (@workers * @threads).times.map { spawn_worker_thread }
106
166
  end
167
+ start_quarantine_monitor if @quarantine_timeout_ms
168
+ Native.control_ready(@id)
169
+ HookFire.fire(@after_boot, "after_boot")
107
170
  @started = true
108
171
  self
109
172
  end
@@ -119,6 +182,7 @@ module Kino
119
182
  def shutdown(timeout: nil)
120
183
  return unless @started
121
184
 
185
+ @quarantine_monitor&.stop
122
186
  deadline = monotonic_now + (timeout || @shutdown_timeout)
123
187
  Native.stop_accepting(@id)
124
188
 
@@ -144,6 +208,9 @@ module Kino
144
208
  end
145
209
 
146
210
  Native.shutdown_runtime(@id, 1_000)
211
+ # The control thread reports "draining" for the whole drain and stops
212
+ # only now, once there is nothing left to report.
213
+ Native.control_stop(@id)
147
214
  # The runtime is gone, so hyper has dropped every pinned buffer;
148
215
  # the keeper (and the strings it marked) may now be collected.
149
216
  @pin_keeper = nil
@@ -159,22 +226,34 @@ module Kino
159
226
  @supervisor ? @supervisor.join : @worker_threads.each(&:join)
160
227
  end
161
228
 
162
- # Production entry point: start, print the banner, trap INT/TERM for
163
- # graceful shutdown (second signal force-exits), block until done.
164
- # The `kino` CLI funnels into this too (CLI#serve).
229
+ # Production entry point: build the server and {#run} it. The `kino`
230
+ # CLI funnels into this too (CLI#serve).
165
231
  #
166
232
  # @param app [#call] a Rack 3 application
167
233
  # @param opts [Hash] see #initialize
168
234
  # @return [Kino::Server] the (stopped) server, after shutdown
169
235
  def self.run(app, **opts)
170
- server = new(app, **opts)
236
+ new(app, **opts).run
237
+ end
238
+
239
+ # Serve until shut down: start, print the banner, trap INT/TERM for
240
+ # graceful shutdown (second signal force-exits), block until done.
241
+ # The Rack handler calls this on a server it built itself.
242
+ #
243
+ # @return [self] after shutdown
244
+ def run
245
+ # Startup output must land immediately even when stdout is a pipe or
246
+ # file (process supervisors, `kino > server.log`, `rails server`
247
+ # under Docker); block buffering would hold the banner back until
248
+ # exit.
249
+ $stdout.sync = true
171
250
  CLI.opening_credits
172
- server.start
173
- CLI.action!(server)
251
+ start
252
+ CLI.action!(self)
174
253
  CLI.fin_at_exit
175
- trap_signals(server)
176
- server.wait
177
- server
254
+ self.class.trap_signals(self)
255
+ wait
256
+ self
178
257
  end
179
258
 
180
259
  # Signal handling shared by Server.run and the kino CLI: INT/TERM drain
@@ -186,14 +265,14 @@ module Kino
186
265
  # kill -USR1 <pid> prints a one-line stats snapshot (find the pid in
187
266
  # the pidfile when configured).
188
267
  trap("USR1") do
189
- Thread.new { $stdout.puts Kino::CLI.stats_line(server.stats) }
268
+ Thread.new { Log.info(CLI.stats_line(server.stats)) }
190
269
  end
191
270
  signaled = false
192
271
  %w[INT TERM].each do |signal|
193
272
  trap(signal) do
194
273
  Process.exit!(1) if signaled
195
274
  signaled = true
196
- $stderr.write("Kino: draining (signal again to force exit)\n")
275
+ Log.warn("draining (signal again to force exit)")
197
276
  # Trap context forbids mutexes; do the real work on a thread.
198
277
  Thread.new { server.shutdown }
199
278
  end
@@ -205,22 +284,90 @@ module Kino
205
284
  #
206
285
  # @return [Hash{Symbol => Object}] mode, lanes, workers, threads,
207
286
  # batch, respawns; plus queued, in_flight, served, rejected,
208
- # timeouts (and lane_depths in lanes mode) once started
287
+ # timeouts, worker_status, quarantined, queue_time (and lane_depths in
288
+ # lanes mode) once started
209
289
  def stats
210
290
  base = {
211
291
  mode: @mode, lanes: @lanes, workers: @workers, threads: @threads,
212
- batch: @batch, respawns: @supervisor ? @supervisor.respawns : 0
292
+ batch: @batch, respawns: 0
213
293
  }
214
294
  return base unless @started
215
295
 
216
- queued, in_flight, served, rejected, timeouts, lane_depths = Native.server_stats(@id)
217
- base.merge!(queued:, in_flight:, served:, rejected:, timeouts:)
296
+ queued, in_flight, served, rejected, timeouts, respawns, lane_depths = Native.server_stats(@id)
297
+ base.merge!(queued:, in_flight:, served:, rejected:, timeouts:, respawns:)
218
298
  base[:lane_depths] = lane_depths if lane_depths
299
+ rows = Native.worker_stats(@id)
300
+ base[:worker_status] = rows.map do |index, served, in_flight, busy_ms, quarantined|
301
+ {index:, served:, in_flight:, busy_ms:, quarantined:}
302
+ end
303
+ base[:quarantined] = rows.count { |_index, _served, _in_flight, _busy_ms, quarantined| quarantined }
304
+ count, sum_seconds = Native.queue_time(@id)
305
+ base[:queue_time] = {count:, sum_seconds:}
219
306
  base
220
307
  end
221
308
 
222
309
  private
223
310
 
311
+ # Register a fresh dispatch slot and run a worker thread on it; returns
312
+ # the thread. Used at boot and by the quarantine replacer.
313
+ def spawn_worker_thread
314
+ worker_id = Native.register_worker(@id)
315
+ Thread.new do
316
+ # Named so log lines from inside say which worker spoke.
317
+ Thread.current.name = "worker-#{worker_id}"
318
+ error = nil
319
+ begin
320
+ Worker.run(@id, worker_id, @app, @batch, @worker_hooks)
321
+ rescue Exception => e # rubocop:disable Lint/RescueException -- a hard crash in a threaded worker thread
322
+ error = e
323
+ raise
324
+ ensure
325
+ HookFire.fire(@on_worker_exit, "on_worker_exit", worker_id, error)
326
+ end
327
+ end
328
+ end
329
+
330
+ # Track a replacement thread spawned outside the initial pool assignment
331
+ # (the quarantine replacer) so shutdown's join/done?/kill sweeps see it.
332
+ def track_replacement_thread(thread)
333
+ @worker_threads_lock.synchronize { @worker_threads << thread }
334
+ end
335
+
336
+ # @private
337
+ # The :threaded-mode quarantine replacer: spawns a replacement worker
338
+ # thread, quarantines the wedged slot, then tracks the new thread so
339
+ # shutdown's join/done?/kill sweeps see it. Built from bound Method
340
+ # objects instead of a server reference, so it drives the server
341
+ # through those methods without send or instance_variable_get.
342
+ class ThreadedReplacer
343
+ def initialize(server_id:, spawner:, tracker:)
344
+ @server_id = server_id
345
+ @spawner = spawner
346
+ @tracker = tracker
347
+ end
348
+
349
+ def replace(worker_id)
350
+ thread = @spawner.call # spawn FIRST (may raise ThreadError)
351
+ Native.quarantine_slot(@server_id, worker_id) # quarantine after success
352
+ @tracker.call(thread)
353
+ true
354
+ end
355
+ end
356
+ private_constant :ThreadedReplacer
357
+
358
+ # A replacer.replace(worker_id) spawns a replacement worker, then
359
+ # quarantines the wedged slot, mode-appropriately. In :ractor the
360
+ # supervisor is the replacer; in :threaded a small object over
361
+ # spawn_worker_thread.
362
+ def start_quarantine_monitor
363
+ replacer = @supervisor || ThreadedReplacer.new(server_id: @id, spawner: method(:spawn_worker_thread),
364
+ tracker: method(:track_replacement_thread))
365
+ @quarantine_monitor = QuarantineMonitor.new(
366
+ server_id: @id, timeout_ms: @quarantine_timeout_ms,
367
+ max: @quarantine_max, replacer: replacer
368
+ ).start
369
+ end
370
+
224
371
  def validate_tls(tls)
225
372
  return nil if tls.nil?
226
373
  unless tls.is_a?(Hash) && tls[:cert] && tls[:key]
@@ -230,10 +377,10 @@ module Kino
230
377
  {cert: String(tls[:cert]), key: String(tls[:key])}
231
378
  end
232
379
 
233
- def validate_on_error(handler)
380
+ def validate_hook(handler, name)
234
381
  return nil if handler.nil?
235
382
  unless handler.respond_to?(:call)
236
- raise ArgumentError, "on_error must respond to #call (got #{handler.class})"
383
+ raise ArgumentError, "#{name} must respond to #call (got #{handler.class})"
237
384
  end
238
385
 
239
386
  handler
@@ -320,7 +467,8 @@ module Kino
320
467
  if @supervisor
321
468
  @supervisor.shutdown([deadline - monotonic_now, 0].max)
322
469
  else
323
- @worker_threads.each do |thread|
470
+ threads = @worker_threads_lock.synchronize { @worker_threads.dup }
471
+ threads.each do |thread|
324
472
  thread.join([deadline - monotonic_now, 0.01].max)
325
473
  end
326
474
  end
@@ -330,7 +478,8 @@ module Kino
330
478
  if @supervisor
331
479
  @supervisor.done?
332
480
  else
333
- @worker_threads.none?(&:alive?)
481
+ threads = @worker_threads_lock.synchronize { @worker_threads.dup }
482
+ threads.none?(&:alive?)
334
483
  end
335
484
  end
336
485
 
@@ -338,9 +487,10 @@ module Kino
338
487
  if @supervisor
339
488
  # Ractors cannot be force-killed; their clients were already freed
340
489
  # by abort_all_inflight. The stuck ractor leaks until process exit.
341
- Native.log_error("shutdown deadline passed with stuck ractor workers") unless @supervisor.done?
490
+ Log.error("shutdown deadline passed with stuck ractor workers") unless @supervisor.done?
342
491
  else
343
- @worker_threads.each { |thread| thread.kill if thread.alive? }
492
+ threads = @worker_threads_lock.synchronize { @worker_threads.dup }
493
+ threads.each { |thread| thread.kill if thread.alive? }
344
494
  end
345
495
  end
346
496
 
@@ -361,18 +511,18 @@ module Kino
361
511
  "Ractor.shareable_proc endpoints); try Ractor.make_shareable(app) " \
362
512
  "or mode: :threaded"
363
513
  end
364
- unless @on_error.nil? || Ractor.shareable?(@on_error)
514
+ if (name = unshareable_worker_hook_name)
365
515
  raise Error,
366
- "mode: :ractor requires a Ractor-shareable on_error handler " \
516
+ "mode: :ractor requires a Ractor-shareable #{name} hook " \
367
517
  "(build it with Ractor.shareable_proc, or use mode: :threaded)"
368
518
  end
369
519
  :ractor
370
520
  when :auto
371
521
  if !Ractor.shareable?(@app)
372
- warn "Kino: app is not Ractor-shareable; falling back to mode: :threaded"
522
+ Log.warn("app is not Ractor-shareable; falling back to mode: :threaded")
373
523
  :threaded
374
- elsif !(@on_error.nil? || Ractor.shareable?(@on_error))
375
- warn "Kino: on_error handler is not Ractor-shareable; falling back to mode: :threaded"
524
+ elsif (name = unshareable_worker_hook_name)
525
+ Log.warn("#{name} hook is not Ractor-shareable; falling back to mode: :threaded")
376
526
  :threaded
377
527
  else
378
528
  :ractor
@@ -381,5 +531,22 @@ module Kino
381
531
  raise ArgumentError, "mode must be :auto, :ractor, or :threaded (got #{requested.inspect})"
382
532
  end
383
533
  end
534
+
535
+ # The hooks that ride into worker context (a ractor in :ractor mode)
536
+ # and so must be Ractor-shareable there. after_boot and on_worker_exit
537
+ # run on the main thread and are exempt.
538
+ def worker_context_hooks
539
+ [[:on_error, @on_error], [:after_worker_boot, @after_worker_boot],
540
+ [:after_request_complete, @after_request_complete]]
541
+ end
542
+
543
+ # The name of the first worker-context hook that is set but not
544
+ # Ractor-shareable, or nil if all set ones are. Used by both the
545
+ # :ractor raise and the :auto warn/fallback branches, so "first
546
+ # offender" is defined once.
547
+ def unshareable_worker_hook_name
548
+ bad = worker_context_hooks.find { |_name, hook| !(hook.nil? || Ractor.shareable?(hook)) }
549
+ bad&.first
550
+ end
384
551
  end
385
552
  end
@@ -8,7 +8,9 @@
8
8
  ## Network
9
9
 
10
10
  # Address to listen on. Use "0.0.0.0" to accept connections from other
11
- # machines.
11
+ # machines, or "unix:///run/kino.sock" to listen on a unix domain socket
12
+ # behind a proxy such as nginx (the port below is then unused; a stale
13
+ # socket file is reclaimed, a live one refused).
12
14
  # bind "127.0.0.1"
13
15
 
14
16
  # Port to listen on.
@@ -22,7 +24,8 @@
22
24
 
23
25
  # How many workers to run. Each worker handles requests independently;
24
26
  # in :ractor mode every worker runs Ruby in parallel on its own core.
25
- # Default: one per CPU core.
27
+ # Default: the CPUs this process may use (Kino.available_parallelism:
28
+ # the affinity mask and, in a container, the cgroup CPU quota).
26
29
  # workers 8
27
30
 
28
31
  # Threads inside each worker. More threads help when your app spends
@@ -74,9 +77,12 @@
74
77
  # for quick handlers; behavior under heavy overload differs slightly.
75
78
  # lanes false
76
79
 
77
- # Print one line per request to stdout, colored by status on a
78
- # terminal. This is the server's view: it includes requests your app
79
- # never saw, such as 503s. Recommended in development.
80
+ # Log every request to stdout: an arrival line before the app runs and a
81
+ # completion line after it, colored by status on a terminal, with a
82
+ # timing breakdown (time in Ruby with its GC pause and allocations, the
83
+ # server's own overhead, and queue wait). This is the server's view: it
84
+ # includes requests your app never saw, such as 503s. Recommended in
85
+ # development; cheap enough for production.
80
86
  # log_requests false
81
87
 
82
88
  # Called when a worker catches an app or delivery error, after the client
@@ -87,6 +93,27 @@
87
93
  # must be Ractor-shareable (build it with Ractor.shareable_proc).
88
94
  # on_error ->(error, env) { ExceptionService.capture(error) }
89
95
 
96
+ # Called once on the main thread after the worker pool is up. Wire
97
+ # readiness here (sd_notify, a "server ready" metric).
98
+ # after_boot { }
99
+
100
+ # Called once inside each worker before it serves, with the worker's slot
101
+ # id. In :ractor mode it runs inside the worker ractor, so it must be
102
+ # Ractor-shareable (build it with Ractor.shareable_proc).
103
+ # after_worker_boot { |worker_id| }
104
+
105
+ # Called inside the worker after each successful response, with (env,
106
+ # status). This is the hot path: leave it unset for zero cost. In :ractor
107
+ # mode it must be Ractor-shareable.
108
+ # after_request_complete { |env, status| }
109
+
110
+ # Called on the main thread when a worker exits, with (worker_index,
111
+ # error). error is the crash cause, or nil on a clean exit. In :ractor
112
+ # mode worker_index identifies the exited ractor (0..workers - 1), a
113
+ # different number space than after_worker_boot's slot id: do not
114
+ # correlate boot and exit by that number in :ractor mode.
115
+ # on_worker_exit { |worker_index, error| }
116
+
90
117
  ## Lifecycle
91
118
 
92
119
  # On shutdown, give in-flight requests this many seconds to finish.
@@ -102,6 +129,36 @@
102
129
  # heavily CPU-bound apps, try 1 to leave more cores for Ruby.
103
130
  # tokio_threads 4
104
131
 
132
+ ## Control plane
133
+
134
+ # Serve read-only monitoring on a separate address: live stats as JSON
135
+ # at /stats, Prometheus text at /metrics, and /ready and /live probes
136
+ # for an orchestrator. Answered by the native layer on its own thread,
137
+ # so it stays live even when every Ruby worker is busy or stuck.
138
+ # Accepts "host:port" or "unix://path". Off unless set.
139
+ # control_bind "127.0.0.1:9293"
140
+
141
+ # When set, /stats and /metrics require "Authorization: Bearer <token>".
142
+ # The probes stay open; they carry no data.
143
+ # control_token ENV["KINO_CONTROL_TOKEN"]
144
+
145
+ ## Quarantine
146
+
147
+ # Quarantine a dispatch slot whose current request has run longer than
148
+ # this many seconds and spawn a replacement to restore capacity. The
149
+ # wedged worker is never interrupted or force-killed; its slot stays
150
+ # quarantined for good (in :ractor mode the ractor itself leaks until
151
+ # process exit, since a wedged ractor cannot be safely interrupted). Off
152
+ # unless set; set it above your slowest legitimate endpoint, and above
153
+ # request_timeout.
154
+ # quarantine_timeout 60
155
+
156
+ # Cap on the total number of replacement events over the process
157
+ # lifetime. Past the cap the monitor stops replacing and the server runs
158
+ # at reduced capacity. Default: the worker count in :ractor mode, workers
159
+ # x threads in :threaded.
160
+ # quarantine_max 8
161
+
105
162
  ## App
106
163
 
107
164
  # Rackup file to load (a command-line argument wins).
data/lib/kino/version.rb CHANGED
@@ -2,5 +2,5 @@
2
2
 
3
3
  module Kino
4
4
  # The gem version (single source of truth; ext/kino/Cargo.toml syncs).
5
- VERSION = "0.2.1"
5
+ VERSION = "0.4.0"
6
6
  end
data/lib/kino/worker.rb CHANGED
@@ -18,13 +18,14 @@ module Kino
18
18
 
19
19
  module_function
20
20
 
21
- def run(server_id, worker_id, app, batch_size = 1, on_error = nil)
21
+ def run(server_id, worker_id, app, batch_size = 1, hooks = nil)
22
+ fire_after_worker_boot(hooks, worker_id)
22
23
  if batch_size <= 1
23
24
  env = Native.take_one(server_id, worker_id)
24
- env = handle_one(env, server_id, worker_id, app, on_error) while env
25
+ env = handle_one(env, server_id, worker_id, app, hooks) while env
25
26
  else
26
27
  batch = Native.take_batch(server_id, worker_id, batch_size)
27
- batch = process(batch, server_id, worker_id, app, batch_size, on_error) while batch
28
+ batch = process(batch, server_id, worker_id, app, batch_size, hooks) while batch
28
29
  end
29
30
  end
30
31
 
@@ -34,8 +35,8 @@ module Kino
34
35
  NOT_FUSED = Object.new.freeze
35
36
 
36
37
  # Handle one request; returns the next env (fused take) or nil.
37
- def handle_one(env, server_id, worker_id, app, on_error)
38
- result = serve(env, app, on_error) do |request, status, headers, chunks|
38
+ def handle_one(env, server_id, worker_id, app, hooks)
39
+ result = serve(env, app, hooks) do |request, status, headers, chunks|
39
40
  request.respond_and_take_one(server_id, worker_id, status, headers, chunks)
40
41
  end
41
42
  result.equal?(NOT_FUSED) ? Native.take_one(server_id, worker_id) : result
@@ -43,10 +44,10 @@ module Kino
43
44
 
44
45
  # Handle every env in the batch; returns the next batch (the last
45
46
  # simple response rides the fused respond_and_take) or nil on shutdown.
46
- def process(batch, server_id, worker_id, app, batch_size, on_error)
47
+ def process(batch, server_id, worker_id, app, batch_size, hooks)
47
48
  last = batch.size - 1
48
49
  batch.each_with_index do |env, index|
49
- result = serve(env, app, on_error) do |request, status, headers, chunks|
50
+ result = serve(env, app, hooks) do |request, status, headers, chunks|
50
51
  if index == last
51
52
  request.respond_and_take(server_id, worker_id, batch_size,
52
53
  status, headers, chunks)
@@ -66,17 +67,39 @@ module Kino
66
67
  # here and return NOT_FUSED. App errors must never kill the worker;
67
68
  # hard crashes (Exception) are the supervisor's job; and `abort` does
68
69
  # the right thing whether or not the response head already went out.
69
- def serve(env, app, on_error)
70
+ def serve(env, app, hooks)
70
71
  request = env[KINO_REQUEST]
71
72
  env[RACK_INPUT] ||= Input.new(request)
72
- status, headers, body = app.call(env)
73
+ if hooks&.access_timing
74
+ # The access log's breakdown: the VM's cumulative GC time and
75
+ # allocation count, differenced around the app call.
76
+ gc_before = GC.total_time
77
+ allocated_before = GC.stat(:total_allocated_objects)
78
+ status, headers, body = app.call(env)
79
+ request.timing(GC.total_time - gc_before, GC.stat(:total_allocated_objects) - allocated_before)
80
+ else
81
+ status, headers, body = app.call(env)
82
+ end
73
83
 
74
84
  if body.respond_to?(:to_ary)
75
- result = yield(request, status.to_i, headers, join_chunks(body.to_ary))
76
- body.close if body.respond_to?(:close)
77
- result
85
+ chunks = join_chunks(body.to_ary)
86
+ if hooks&.after_request_complete
87
+ # Hook set: do not fuse. Send the complete response, fire the hook
88
+ # after it is out, then signal the caller to take the next request
89
+ # separately (so the hook never waits on the next request).
90
+ request.send_simple(status.to_i, headers, chunks)
91
+ body.close if body.respond_to?(:close)
92
+ fire_after_request_complete(hooks, env, status.to_i)
93
+ NOT_FUSED
94
+ else
95
+ # No hook: fused fast path, unchanged, zero cost.
96
+ result = yield(request, status.to_i, headers, chunks)
97
+ body.close if body.respond_to?(:close)
98
+ result
99
+ end
78
100
  else
79
101
  deliver_streaming(request, status.to_i, headers, body, env[RACK_INPUT])
102
+ fire_after_request_complete(hooks, env, status.to_i)
80
103
  NOT_FUSED
81
104
  end
82
105
  rescue => e
@@ -85,27 +108,12 @@ module Kino
85
108
  # delivery errors (they happen after app.call returned, so no
86
109
  # middleware can see them); its own failures are logged, not raised,
87
110
  # because nothing may escape this block and kill the worker.
88
- Native.log_error(error_log_line(e))
111
+ Log.exception(e, env)
89
112
  request.abort
90
- if on_error
91
- begin
92
- on_error.call(e, env)
93
- rescue => hook_error
94
- Native.log_error("on_error hook raised #{hook_error.class}: #{hook_error.message}")
95
- end
96
- end
113
+ HookFire.fire(hooks&.on_error, "on_error", e, env)
97
114
  NOT_FUSED
98
115
  end
99
116
 
100
- # First frames only: the raise site is at the top, and Rails stacks
101
- # run hundreds of middleware frames deep. Hooks get the full exception.
102
- BACKTRACE_FRAMES = 12
103
-
104
- def error_log_line(error)
105
- ["#{error.class}: #{error.message}",
106
- *(error.backtrace || []).first(BACKTRACE_FRAMES)].join("\n ")
107
- end
108
-
109
117
  def deliver_streaming(request, status, headers, body, input)
110
118
  request.send_headers(status, headers)
111
119
  if body.respond_to?(:call) && !body.respond_to?(:each)
@@ -142,7 +150,15 @@ module Kino
142
150
  joined
143
151
  end
144
152
 
153
+ def fire_after_worker_boot(hooks, worker_id)
154
+ HookFire.fire(hooks&.after_worker_boot, "after_worker_boot", worker_id)
155
+ end
156
+
157
+ def fire_after_request_complete(hooks, env, status)
158
+ HookFire.fire(hooks&.after_request_complete, "after_request_complete", env, status)
159
+ end
160
+
145
161
  private_class_method :handle_one, :process, :serve, :deliver_streaming,
146
- :join_chunks, :error_log_line
162
+ :join_chunks, :fire_after_worker_boot, :fire_after_request_complete
147
163
  end
148
164
  end
@@ -0,0 +1,13 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Kino
4
+ # @private
5
+ # The worker-context lifecycle hooks, bundled so one frozen, shareable
6
+ # value crosses into each worker (a ractor in :ractor mode) instead of
7
+ # several bare procs. Any member may be nil. A Data instance is frozen,
8
+ # so it is Ractor.shareable? exactly when its members are (nil, or a
9
+ # Ractor.shareable_proc), letting it ride the ractor boundary like the app.
10
+ # `access_timing` rides along: whether the worker measures the GC pause
11
+ # and allocations around each app call for the access log's breakdown.
12
+ WorkerHooks = Data.define(:on_error, :after_worker_boot, :after_request_complete, :access_timing)
13
+ end
data/lib/kino.rb CHANGED
@@ -33,9 +33,20 @@ module Kino
33
33
  remaining = Native.sleep_chunk(remaining) while remaining.positive?
34
34
  nil
35
35
  end
36
+
37
+ # How many CPUs this process may actually use: the `workers` default.
38
+ # Unlike `Etc.nprocessors`, this honours a cgroup CPU quota (a container
39
+ # limited to 2 CPUs on a 64-core host gets 2, not 64) as well as the
40
+ # affinity mask; a fractional quota rounds up. Never below 1.
41
+ #
42
+ # @return [Integer]
43
+ def self.available_parallelism
44
+ Native.available_parallelism
45
+ end
36
46
  end
37
47
 
38
48
  require_relative "kino/cli"
49
+ require_relative "kino/log"
39
50
  require_relative "kino/logger"
40
51
  require_relative "kino/check"
41
52
  require_relative "kino/input"
@@ -43,8 +54,11 @@ require_relative "kino/null_input"
43
54
  require_relative "kino/errors_stream"
44
55
  require_relative "kino/stream"
45
56
  require_relative "kino/configuration"
57
+ require_relative "kino/hook_fire"
58
+ require_relative "kino/worker_hooks"
46
59
  require_relative "kino/worker"
47
60
  require_relative "kino/ractor_supervisor"
61
+ require_relative "kino/quarantine_monitor"
48
62
  require_relative "kino/server"
49
63
 
50
64
  # Hand the frozen shareable singletons to the native layer: it sets them