kino 0.2.1-aarch64-linux → 0.4.0-aarch64-linux

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/lib/kino/log.rb ADDED
@@ -0,0 +1,104 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Kino
4
+ # Server log lines: the lifecycle notices, crash and respawn reports,
5
+ # hook failures, the failed-request report, and whatever apps write to
6
+ # `rack.errors`, all in one shape:
7
+ #
8
+ # kino[4213] worker-3: after_worker_boot hook raised RuntimeError: boom
9
+ #
10
+ # The label is syslog's `ident[pid]` tag plus the source that spoke: the
11
+ # worker ractor and/or thread by name, `main` for neither. On color
12
+ # terminals the label is dim, yellow, or red by level; the message stays
13
+ # plain. Notes go to stdout, warnings and errors to stderr.
14
+ #
15
+ # Hooks may log through here too (`Kino::Log.info "cache warm"`). Every
16
+ # method is safe inside a worker ractor: the line is handed to the
17
+ # native layer, which owns the streams, so no ractor touches `$stdout`
18
+ # or `$stderr` itself.
19
+ module Log
20
+ # Frames shown in a failed-request report before the rest are folded.
21
+ FRAMES = 12
22
+
23
+ # The working directory at boot, stripped from backtrace frames so the
24
+ # app's own code reads `app/controllers/x.rb:9` rather than an
25
+ # absolute path (frozen: worker ractors read it).
26
+ WORKING_DIR = File.join(Dir.pwd, "").freeze
27
+
28
+ module_function
29
+
30
+ # @param message [#to_s]
31
+ # @return [void]
32
+ def info(message)
33
+ Native.log_line("info", source, message.to_s)
34
+ end
35
+
36
+ # @param message [#to_s]
37
+ # @return [void]
38
+ def warn(message)
39
+ Native.log_line("warn", source, message.to_s)
40
+ end
41
+
42
+ # @param message [#to_s]
43
+ # @return [void]
44
+ def error(message)
45
+ Native.log_line("error", source, message.to_s)
46
+ end
47
+
48
+ # The failed-request report: the request line, the error, and where it
49
+ # raised in the app, then the backtrace with the app's own frames
50
+ # first (relative to the working directory) and the rest folded.
51
+ #
52
+ # 500 GET /boom · RuntimeError: kaboom (app.rb:12:in 'explode')
53
+ # app.rb:12:in 'explode'
54
+ # /gems/rack-3.2.7/lib/rack/builder.rb:...
55
+ # … 38 more
56
+ #
57
+ # @param error [Exception]
58
+ # @param env [Hash] the Rack env of the failed request
59
+ # @param status [Integer] the status the client got
60
+ # @return [void]
61
+ def exception(error, env, status: 500)
62
+ frames, depth = trace(error)
63
+ site = frames.first ? " (#{frames.first})" : ""
64
+ lines = ["#{status} #{env["REQUEST_METHOD"]} #{env["PATH_INFO"]} · #{error.class}: #{error.message}#{site}"]
65
+ frames.each { |frame| lines << " #{frame}" }
66
+ lines << " … #{depth - frames.size} more" if depth > frames.size
67
+ error(lines.join("\n"))
68
+ end
69
+
70
+ # The `kino[<pid>] <source>:` tag a line from here carries.
71
+ # @return [String]
72
+ def label
73
+ "kino[#{Process.pid}] #{source}:"
74
+ end
75
+
76
+ # Who is speaking: the ractor's name, the thread's name, both joined
77
+ # with a slash, or `main` when neither is named. Kino names its
78
+ # worker ractors and threads `worker-N`.
79
+ # @return [String]
80
+ def source
81
+ parts = [Ractor.current.name, Thread.current.name].compact
82
+ parts.empty? ? "main" : parts.join("/")
83
+ end
84
+
85
+ # The backtrace as [frames, depth]: each frame relativized to the
86
+ # working directory, the app's own frames floated to the front (the
87
+ # raise site in your code reads first; gem and stdlib frames keep
88
+ # their order below), capped at FRAMES; depth is the real length.
89
+ def trace(error)
90
+ raw = error.backtrace || []
91
+ app, rest = raw.map { |frame| frame.delete_prefix(WORKING_DIR) }.partition { |frame| app_frame?(frame) }
92
+ [(app + rest).first(FRAMES), raw.size]
93
+ end
94
+
95
+ # A project-relative path (the working-directory prefix came off, so
96
+ # it does not start with `/`) that is not a synthetic frame (`(eval)`,
97
+ # `<internal:...>`). Gem and stdlib frames stay absolute.
98
+ def app_frame?(frame)
99
+ !frame.start_with?("/", "<", "(")
100
+ end
101
+
102
+ private_class_method :trace, :app_frame?
103
+ end
104
+ end
@@ -0,0 +1,73 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Kino
4
+ # @private
5
+ # Polls per-slot busy_ms and, past the timeout, quarantines a wedged slot
6
+ # and asks the replacer to spawn a fresh worker. Runs one thread on the
7
+ # main ractor (uncontended by wedged worker ractors, so it stays
8
+ # responsive in :ractor mode). Never interrupts the wedged worker.
9
+ class QuarantineMonitor
10
+ def initialize(server_id:, timeout_ms:, max:, replacer:, tick: 0.5)
11
+ @server_id = server_id
12
+ @timeout_ms = timeout_ms
13
+ @max = max
14
+ @replacer = replacer
15
+ @tick = tick
16
+ @outstanding = 0
17
+ @at_cap_logged = false
18
+ @running = false
19
+ @thread = nil
20
+ end
21
+
22
+ def start
23
+ @running = true
24
+ @thread = Thread.new do
25
+ Thread.current.name = "quarantine"
26
+ run
27
+ end
28
+ self
29
+ end
30
+
31
+ def stop
32
+ @running = false
33
+ @thread&.join(@tick * 2)
34
+ end
35
+
36
+ private
37
+
38
+ def run
39
+ tick while @running
40
+ rescue => e
41
+ Log.error("quarantine monitor crashed: #{e.class}: #{e.message}")
42
+ end
43
+
44
+ def tick
45
+ scan_slots
46
+ rescue => e
47
+ # A bad tick must never kill the monitor.
48
+ Log.error("quarantine tick error: #{e.class}: #{e.message}")
49
+ ensure
50
+ sleep @tick
51
+ end
52
+
53
+ def scan_slots
54
+ Native.worker_stats(@server_id).each do |index, _served, _in_flight, busy_ms, quarantined|
55
+ next if quarantined || busy_ms <= @timeout_ms
56
+
57
+ if @outstanding >= @max
58
+ unless @at_cap_logged
59
+ Log.warn("quarantine at cap (#{@max}); serving at reduced capacity")
60
+ @at_cap_logged = true
61
+ end
62
+ next
63
+ end
64
+
65
+ if @replacer.replace(index)
66
+ Native.record_quarantine_replacement(@server_id)
67
+ @outstanding += 1
68
+ @at_cap_logged = false
69
+ end
70
+ end
71
+ end
72
+ end
73
+ end
@@ -7,19 +7,23 @@ module Kino
7
7
  # ractor, Exception from app code included) wakes it to 500 the in-flight
8
8
  # requests and respawn. Clean exits (queue drained) end supervision.
9
9
  class RactorSupervisor
10
- attr_reader :respawns
11
-
12
- def initialize(server_id, app, workers:, threads:, batch: 1, on_error: nil)
10
+ def initialize(server_id, app, workers:, threads:, batch: 1, hooks: nil, on_worker_exit: nil)
13
11
  @server_id = server_id
14
12
  @app = app
15
13
  @workers = workers
16
14
  @threads = threads
17
15
  @batch = batch
18
- @on_error = on_error
19
- @respawns = 0
16
+ @hooks = hooks
17
+ @on_worker_exit = on_worker_exit
20
18
  @draining = false
21
19
  @lock = Mutex.new
22
20
  @supervisor_threads = []
21
+ @worker_slots = {}
22
+ @slot_to_worker = {}
23
+ @replaced = {}
24
+ # The first replacement's index; `replace` increments before using it,
25
+ # so this starts one below the first free index (@workers).
26
+ @next_worker_index = @workers - 1
23
27
  end
24
28
 
25
29
  def start
@@ -32,43 +36,81 @@ module Kino
32
36
  def shutdown(timeout)
33
37
  @lock.synchronize { @draining = true }
34
38
  deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + timeout
35
- @supervisor_threads.each do |thread|
39
+ @lock.synchronize { @supervisor_threads.dup }.each do |thread|
36
40
  remaining = deadline - Process.clock_gettime(Process::CLOCK_MONOTONIC)
37
41
  thread.join([remaining, 0.01].max)
38
42
  end
39
43
  end
40
44
 
41
45
  def done?
42
- @supervisor_threads.none?(&:alive?)
46
+ @lock.synchronize { @supervisor_threads.dup }.none?(&:alive?)
43
47
  end
44
48
 
45
49
  # Block until the workers exit on their own (drain elsewhere): join
46
50
  # without flipping the draining flag.
47
51
  def join
48
- @supervisor_threads.each(&:join)
52
+ @lock.synchronize { @supervisor_threads.dup }.each(&:join)
53
+ end
54
+
55
+ # Replace the ractor owning slot `worker_id`: spawn a fresh supervised
56
+ # ractor, then quarantine the old ractor's slots. The old supervisor
57
+ # thread stays blocked in ractor.value on the wedged ractor (it and the
58
+ # ractor leak until process exit; a wedged ractor cannot be
59
+ # force-killed). Returns true if a replacement was spawned.
60
+ def replace(worker_id)
61
+ worker_index = @lock.synchronize { @slot_to_worker[worker_id] }
62
+ return false unless worker_index
63
+
64
+ # Idempotent per ractor: a stale monitor snapshot can list two sibling
65
+ # slots of the same ractor, only the first replaces it.
66
+ claimed = @lock.synchronize do
67
+ if @replaced.key?(worker_index)
68
+ false
69
+ else
70
+ @replaced[worker_index] = true
71
+ true
72
+ end
73
+ end
74
+ return false unless claimed
75
+
76
+ new_index = @lock.synchronize { @next_worker_index += 1 }
77
+ thread =
78
+ begin
79
+ supervise(new_index) # spawn FIRST, nothing quarantined yet
80
+ rescue
81
+ @lock.synchronize { @replaced.delete(worker_index) } # allow retry next tick
82
+ raise
83
+ end
84
+ slot_ids = @lock.synchronize { @worker_slots[worker_index] } || []
85
+ slot_ids.each { |id| Native.quarantine_slot(@server_id, id) } # quarantine only after success
86
+ @lock.synchronize { @supervisor_threads << thread }
87
+ true
49
88
  end
50
89
 
51
90
  private
52
91
 
53
92
  def supervise(index)
54
93
  Thread.new do
94
+ Thread.current.name = "supervisor-#{index}"
55
95
  crashes = 0
56
96
  loop do
57
- ractor, worker_ids = spawn_worker
97
+ ractor, worker_ids = spawn_worker(index)
58
98
  begin
59
99
  ractor.value # blocks until the ractor terminates
100
+ HookFire.fire(@on_worker_exit, "on_worker_exit", index, nil) # clean exit: queue drained
60
101
  break # clean exit: queue closed, workers drained
61
102
  rescue Ractor::Error => e
62
103
  # The ractor died mid-flight. Anything it was serving will never
63
104
  # be answered by Ruby: 500 those clients NOW (not when GC gets
64
105
  # around to dropping the dead heap), then decide on respawn.
65
106
  worker_ids.each { |id| Native.abort_inflight(@server_id, id) }
107
+ cause = (e.respond_to?(:cause) && e.cause) ? e.cause : e
108
+ HookFire.fire(@on_worker_exit, "on_worker_exit", index, cause)
66
109
  break if draining?
67
110
 
68
111
  crashes += 1
69
- @lock.synchronize { @respawns += 1 }
70
- cause = (e.respond_to?(:cause) && e.cause) ? e.cause : e
71
- Native.log_error("worker ractor #{index} crashed (#{cause.class}: #{cause.message}); respawning")
112
+ Native.record_respawn(@server_id)
113
+ Log.error("worker-#{index} crashed (#{cause.class}: #{cause.message}); respawning")
72
114
  # Policy (crash recovery): unlimited respawn
73
115
  # keeps the server up under rare crashes but turns a
74
116
  # crash-on-every-request bug into a busy loop. A circuit breaker
@@ -82,15 +124,23 @@ module Kino
82
124
  # Fresh ractor + fresh native slots. Slots are never reused across
83
125
  # respawns: stale interrupt kicks and dead weak refs go down with the
84
126
  # old slot.
85
- def spawn_worker
127
+ def spawn_worker(worker_index)
86
128
  worker_ids = Array.new(@threads) { Native.register_worker(@server_id) }
87
- ractor = Ractor.new(@server_id, worker_ids, @app, @batch, @on_error) do |server_id, ids, app, batch, on_error|
88
- ids.map do |id|
129
+ @lock.synchronize do
130
+ @worker_slots[worker_index] = worker_ids
131
+ worker_ids.each { |id| @slot_to_worker[id] = worker_index }
132
+ end
133
+ # Named so log lines from inside say which worker spoke: the ractor
134
+ # alone for a single thread, `worker-N/thread-M` for more.
135
+ ractor = Ractor.new(@server_id, worker_ids, @app, @batch, @hooks,
136
+ name: "worker-#{worker_index}") do |server_id, ids, app, batch, hooks|
137
+ ids.each_with_index.map do |id, position|
89
138
  Thread.new do
90
139
  # Crashes surface via Ractor#value in the supervisor; don't also
91
140
  # spray the backtrace to stderr from inside the dying ractor.
92
141
  Thread.current.report_on_exception = false
93
- Kino::Worker.run(server_id, id, app, batch, on_error)
142
+ Thread.current.name = "thread-#{position + 1}" if ids.size > 1
143
+ Kino::Worker.run(server_id, id, app, batch, hooks)
94
144
  end
95
145
  end.each(&:join)
96
146
  end