gritz-core 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +13 -0
- data/README.md +3 -1
- data/lib/gritz/cli.rb +62 -36
- data/lib/gritz/configuration.rb +30 -14
- data/lib/gritz/context.rb +31 -2
- data/lib/gritz/controller.rb +14 -0
- data/lib/gritz/core/version.rb +1 -1
- data/lib/gritz/dispatcher.rb +9 -3
- data/lib/gritz/fork_guard.rb +75 -0
- data/lib/gritz/metrics/aggregator.rb +119 -0
- data/lib/gritz/metrics/recorder.rb +89 -0
- data/lib/gritz/middleware/exception_mapper.rb +1 -3
- data/lib/gritz/middleware/logging.rb +27 -3
- data/lib/gritz/middleware/metrics.rb +29 -0
- data/lib/gritz/middleware/stack.rb +1 -1
- data/lib/gritz/supervisor/admin_server.rb +121 -0
- data/lib/gritz/supervisor/launcher.rb +381 -0
- data/lib/gritz/supervisor/master.rb +426 -0
- data/lib/gritz/supervisor/signal_queue.rb +49 -0
- data/lib/gritz/supervisor/status_channel.rb +124 -0
- data/lib/gritz/supervisor/worker_handle.rb +35 -0
- data/lib/gritz/testing/cluster.rb +186 -0
- data/lib/gritz/worker/runner.rb +203 -0
- metadata +34 -2
|
@@ -0,0 +1,426 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "logger"
|
|
4
|
+
|
|
5
|
+
module Gritz
|
|
6
|
+
module Supervisor
|
|
7
|
+
# Forks workers, monitors their status pipes, and owns every child until reaped.
|
|
8
|
+
# @api public
|
|
9
|
+
class Master
|
|
10
|
+
SIGNALS = %w[TERM INT QUIT TTIN TTOU HUP CHLD USR1 USR2].freeze
|
|
11
|
+
attr_reader :workers
|
|
12
|
+
|
|
13
|
+
def initialize(config, logger: Logger.new($stdout), status_io: nil, owner_channel: nil)
|
|
14
|
+
@config = config
|
|
15
|
+
@logger = logger
|
|
16
|
+
@workers = {}
|
|
17
|
+
@desired = config.workers
|
|
18
|
+
@owner_channel = owner_channel
|
|
19
|
+
@reports = owner_channel || (StatusChannel.new(status_io) if status_io)
|
|
20
|
+
@metrics = Metrics::Aggregator.new
|
|
21
|
+
@forwarded = []
|
|
22
|
+
@replacement_queue = []
|
|
23
|
+
@exit_status = 0
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
def run
|
|
27
|
+
@config.validate_runtime!
|
|
28
|
+
raise ConfigurationError, "Supervisor requires workers > 0" unless @desired.positive?
|
|
29
|
+
|
|
30
|
+
require "gritz/native"
|
|
31
|
+
@signals = SignalQueue.new(signals: SIGNALS)
|
|
32
|
+
unless @owner_channel
|
|
33
|
+
@admin = AdminServer.new(bind: @config.admin_bind, status: -> { status }, ready: -> { ready? },
|
|
34
|
+
metrics: -> { @metrics.render(workers: @workers.values) }, logger: @logger)
|
|
35
|
+
end
|
|
36
|
+
@guard = ForkGuard.activate(mode: @config.fork_mode == :clean ? @config.fork_guard : :off, logger: @logger)
|
|
37
|
+
@config.preload! if @config.preload_app?
|
|
38
|
+
Process.warmup if @config.preload_app? && Process.respond_to?(:warmup)
|
|
39
|
+
if @config.workers > 1 && !RUBY_PLATFORM.include?("linux")
|
|
40
|
+
@logger.warn("Multiple native workers require Linux for SO_REUSEPORT load balancing; use workers 0 on macOS")
|
|
41
|
+
end
|
|
42
|
+
maintain_worker_count
|
|
43
|
+
loop do
|
|
44
|
+
@signals.drain.each { |signal| handle_signal(signal) }
|
|
45
|
+
read_statuses
|
|
46
|
+
reap_children
|
|
47
|
+
check_timeouts
|
|
48
|
+
sample_memory
|
|
49
|
+
check_recycle
|
|
50
|
+
advance_replacement
|
|
51
|
+
maintain_worker_count unless @shutdown_at
|
|
52
|
+
flush_forwarded
|
|
53
|
+
@reports&.write(@owner_channel ? status.merge(type: "status") : status) if @forwarded.empty?
|
|
54
|
+
@admin&.poll
|
|
55
|
+
if @owner_channel&.closed?
|
|
56
|
+
@exit_status = 1
|
|
57
|
+
begin_shutdown(immediate: true)
|
|
58
|
+
end
|
|
59
|
+
if @shutdown_at && @workers.empty?
|
|
60
|
+
break if !@owner_channel || @owner_channel.closed? || (@forwarded.empty? && @owner_channel.flush)
|
|
61
|
+
break if now >= @shutdown_at + @config.drain_delay + @config.shutdown_timeout
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
worker_ios = @forwarded.empty? ? @workers.values.filter_map { |handle| handle.channel.io unless handle.channel.closed? } : []
|
|
65
|
+
owner_ios = !@forwarded.empty? && @owner_channel && !@owner_channel.closed? ? [@owner_channel.io] : nil
|
|
66
|
+
IO.select([@signals.io, *@admin&.ios.to_a, *worker_ios], owner_ios, nil, 0.05)
|
|
67
|
+
end
|
|
68
|
+
@exit_status
|
|
69
|
+
ensure
|
|
70
|
+
cleanup
|
|
71
|
+
ForkGuard.deactivate if @guard
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
def status
|
|
75
|
+
{ pid: Process.pid, state: @shutdown_at ? "draining" : "running", desired: @desired,
|
|
76
|
+
phased_restart: !@replacement.nil? || !@replacement_queue.empty?, workers: @workers.values.map(&:to_h) }
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
def ready?
|
|
80
|
+
!@shutdown_at && @workers.values.count { |handle|
|
|
81
|
+
handle.state == "ready" && !handle.term_at && handle.stats[:healthy] != false
|
|
82
|
+
} >= @config.min_ready_workers
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
private
|
|
86
|
+
|
|
87
|
+
def now = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
88
|
+
|
|
89
|
+
def maintain_worker_count
|
|
90
|
+
active = @workers.values.reject(&:term_at)
|
|
91
|
+
used = active.map(&:index)
|
|
92
|
+
(@desired - active.size).times do
|
|
93
|
+
index = (0...@desired).find { |candidate| !used.include?(candidate) }
|
|
94
|
+
spawn_worker(index)
|
|
95
|
+
used << index
|
|
96
|
+
end
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
def spawn_worker(index)
|
|
100
|
+
reader, writer = IO.pipe
|
|
101
|
+
@config.run_hooks(:before_fork, index)
|
|
102
|
+
experimental = @config.fork_mode == :grpc_fork_support
|
|
103
|
+
if experimental
|
|
104
|
+
require "gritz/native"
|
|
105
|
+
Transport::Native.prefork
|
|
106
|
+
prepared = true
|
|
107
|
+
end
|
|
108
|
+
pid = Process.fork do
|
|
109
|
+
exit_status = 1
|
|
110
|
+
begin
|
|
111
|
+
reader.close
|
|
112
|
+
# Restore all inherited master traps, including CHLD and resize signals.
|
|
113
|
+
@signals.close
|
|
114
|
+
@reports&.close
|
|
115
|
+
@admin&.close
|
|
116
|
+
@workers.each_value(&:close)
|
|
117
|
+
Transport::Native.postfork_child if experimental
|
|
118
|
+
exit_status = Worker::Runner.new(index: index, status_io: writer, config: @config, logger: @logger).run
|
|
119
|
+
rescue StandardError, LoadError, SyntaxError, SystemExit => e
|
|
120
|
+
@logger.error("Worker #{index} failed: #{e.full_message}")
|
|
121
|
+
exit_status = 1
|
|
122
|
+
ensure
|
|
123
|
+
begin
|
|
124
|
+
writer.close unless writer.closed?
|
|
125
|
+
ensure
|
|
126
|
+
# Never unwind the inherited master ensure or run application at_exit hooks.
|
|
127
|
+
Process.exit!(exit_status)
|
|
128
|
+
end
|
|
129
|
+
end
|
|
130
|
+
end
|
|
131
|
+
writer.close
|
|
132
|
+
handle = WorkerHandle.new(pid: pid, index: index, status_io: reader)
|
|
133
|
+
handle.recycle_factor = 1.0 + (rand * @config.worker_recycle.fetch(:jitter, 0.0))
|
|
134
|
+
@workers[pid] = handle
|
|
135
|
+
@logger.info("Worker #{index} spawned pid=#{pid}")
|
|
136
|
+
handle
|
|
137
|
+
rescue StandardError
|
|
138
|
+
reader&.close unless reader&.closed?
|
|
139
|
+
writer&.close unless writer&.closed?
|
|
140
|
+
raise
|
|
141
|
+
ensure
|
|
142
|
+
Transport::Native.postfork_parent if prepared
|
|
143
|
+
end
|
|
144
|
+
|
|
145
|
+
def read_statuses
|
|
146
|
+
@workers.each_value do |handle|
|
|
147
|
+
break unless @forwarded.empty?
|
|
148
|
+
|
|
149
|
+
consume_status(handle, handle.channel.read)
|
|
150
|
+
flush_forwarded
|
|
151
|
+
end
|
|
152
|
+
end
|
|
153
|
+
|
|
154
|
+
def consume_status(handle, rows)
|
|
155
|
+
rows.each do |message|
|
|
156
|
+
if message[:type] == "metrics"
|
|
157
|
+
if @metrics.apply(handle, message) && @owner_channel
|
|
158
|
+
@forwarded << message.merge(pid: Process.pid, worker_pid: handle.pid, worker_started_at: handle.born_at)
|
|
159
|
+
end
|
|
160
|
+
next
|
|
161
|
+
end
|
|
162
|
+
handle.update(message, now: now)
|
|
163
|
+
if message[:state] == "failed" && !@ever_ready
|
|
164
|
+
@exit_status = 1
|
|
165
|
+
begin_shutdown
|
|
166
|
+
end
|
|
167
|
+
@ever_ready = true if message[:state] == "ready"
|
|
168
|
+
rescue ConfigurationError => e
|
|
169
|
+
@logger.error("Worker #{handle.pid}: #{e.message}")
|
|
170
|
+
kill(handle)
|
|
171
|
+
end
|
|
172
|
+
end
|
|
173
|
+
|
|
174
|
+
def flush_forwarded
|
|
175
|
+
return unless @owner_channel
|
|
176
|
+
|
|
177
|
+
@owner_channel.flush
|
|
178
|
+
@forwarded.shift while !@forwarded.empty? && @owner_channel.write(@forwarded.first)
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
def reap_children
|
|
182
|
+
# Keep a reaped worker's bounded pipe tail queued until owner backpressure clears.
|
|
183
|
+
return unless @forwarded.empty? || @owner_channel&.closed?
|
|
184
|
+
|
|
185
|
+
# Only reap owned children: application hooks can start unrelated subprocesses.
|
|
186
|
+
@workers.each_key do |pid|
|
|
187
|
+
result = Process.waitpid2(pid, Process::WNOHANG)
|
|
188
|
+
next unless result
|
|
189
|
+
|
|
190
|
+
handle = @workers.fetch(pid)
|
|
191
|
+
# A forked application subprocess can retain the writer after the worker exits.
|
|
192
|
+
# Drain the bytes present at reap time without waiting for that subprocess's EOF.
|
|
193
|
+
available = 0
|
|
194
|
+
unless handle.channel.closed?
|
|
195
|
+
size = [0].pack("i")
|
|
196
|
+
# FIONREAD reports queued bytes; Ruby 4.0 removed IO#nread.
|
|
197
|
+
handle.channel.io.ioctl(RUBY_PLATFORM.include?("linux") ? 0x541B : 0x4004667F, size)
|
|
198
|
+
available = size.unpack1("i")
|
|
199
|
+
end
|
|
200
|
+
if available.zero?
|
|
201
|
+
consume_status(handle, handle.channel.read)
|
|
202
|
+
end
|
|
203
|
+
while available.positive?
|
|
204
|
+
bytes = [available, StatusChannel::MAX_READ_BYTES].min
|
|
205
|
+
consume_status(handle, handle.channel.read(max_bytes: bytes))
|
|
206
|
+
available -= bytes
|
|
207
|
+
break if handle.channel.closed?
|
|
208
|
+
end
|
|
209
|
+
@workers.delete(pid)
|
|
210
|
+
@metrics.forget(handle)
|
|
211
|
+
handle.close
|
|
212
|
+
child_status = result.last
|
|
213
|
+
@logger.info("Worker #{handle.index} exited pid=#{pid} status=#{child_status}")
|
|
214
|
+
startup_failed = (child_status.exited? || child_status.termsig != Signal.list.fetch("KILL")) && !@ever_ready && !@shutdown_at && !handle.term_at
|
|
215
|
+
shutdown_failed = @shutdown_at && !child_status.success?
|
|
216
|
+
if handle.state != "killed" && (startup_failed || shutdown_failed)
|
|
217
|
+
@exit_status = 1
|
|
218
|
+
begin_shutdown(immediate: true) if startup_failed
|
|
219
|
+
end
|
|
220
|
+
if !@shutdown_at && !handle.term_at && @workers.values.count { |worker| !worker.term_at } < @desired
|
|
221
|
+
record_restart(handle.restart_reason || "worker_exit")
|
|
222
|
+
end
|
|
223
|
+
break unless @forwarded.empty?
|
|
224
|
+
rescue Errno::ECHILD
|
|
225
|
+
@workers.delete(pid)&.close
|
|
226
|
+
end
|
|
227
|
+
end
|
|
228
|
+
|
|
229
|
+
def handle_signal(signal)
|
|
230
|
+
case signal
|
|
231
|
+
when "TERM", "INT" then begin_shutdown
|
|
232
|
+
when "QUIT" then begin_shutdown(immediate: true)
|
|
233
|
+
when "TTIN"
|
|
234
|
+
if @replacement || !@replacement_queue.empty?
|
|
235
|
+
@logger.warn("Wait for phased restart before resizing workers")
|
|
236
|
+
elsif !@shutdown_at && @config.bind.end_with?(":0")
|
|
237
|
+
@logger.warn("Cannot add a reuseport worker with port 0; configure a fixed bind port")
|
|
238
|
+
elsif !@shutdown_at
|
|
239
|
+
@desired += 1
|
|
240
|
+
end
|
|
241
|
+
when "TTOU"
|
|
242
|
+
if @replacement || !@replacement_queue.empty?
|
|
243
|
+
@logger.warn("Wait for phased restart before resizing workers")
|
|
244
|
+
elsif !@shutdown_at && @desired > 1
|
|
245
|
+
@desired -= 1
|
|
246
|
+
retire(@workers.values.reject(&:term_at).max_by(&:index), delay: 0)
|
|
247
|
+
end
|
|
248
|
+
when "HUP"
|
|
249
|
+
@logger.reopen
|
|
250
|
+
@workers.each_key { |pid| send_signal("HUP", pid) }
|
|
251
|
+
@logger.info("Log reopened")
|
|
252
|
+
when "USR1"
|
|
253
|
+
if fixed_port? && !@shutdown_at && !@replacement && @replacement_queue.empty?
|
|
254
|
+
@replacement_queue = @workers.values.reject(&:term_at).sort_by(&:index).map { |handle| [handle.pid, "phased_restart"] }
|
|
255
|
+
end
|
|
256
|
+
when "USR2"
|
|
257
|
+
if fixed_port? && !@shutdown_at
|
|
258
|
+
if @owner_channel
|
|
259
|
+
@forwarded << { type: "reexec", pid: Process.pid } unless @forwarded.any? { |row| row[:type] == "reexec" }
|
|
260
|
+
else
|
|
261
|
+
@logger.warn("USR2 requires the gritz launcher")
|
|
262
|
+
end
|
|
263
|
+
end
|
|
264
|
+
end
|
|
265
|
+
end
|
|
266
|
+
|
|
267
|
+
def begin_shutdown(immediate: false)
|
|
268
|
+
@shutdown_at ||= now
|
|
269
|
+
@replacement_queue.clear
|
|
270
|
+
@replacement = nil
|
|
271
|
+
if immediate
|
|
272
|
+
@workers.each_value { |handle| kill(handle) }
|
|
273
|
+
else
|
|
274
|
+
@workers.each_value { |handle| retire(handle) unless handle.term_at }
|
|
275
|
+
end
|
|
276
|
+
end
|
|
277
|
+
|
|
278
|
+
def retire(handle, delay: @config.drain_delay)
|
|
279
|
+
return unless handle
|
|
280
|
+
|
|
281
|
+
handle.term_at = now + delay
|
|
282
|
+
handle.state = "draining"
|
|
283
|
+
send_signal("USR1", handle.pid)
|
|
284
|
+
end
|
|
285
|
+
|
|
286
|
+
def fixed_port?
|
|
287
|
+
return true unless @config.bind.end_with?(":0")
|
|
288
|
+
|
|
289
|
+
@logger.warn("Worker replacement requires a fixed bind port")
|
|
290
|
+
false
|
|
291
|
+
end
|
|
292
|
+
|
|
293
|
+
def advance_replacement
|
|
294
|
+
return if @shutdown_at || !@forwarded.empty?
|
|
295
|
+
|
|
296
|
+
if @replacement
|
|
297
|
+
old = @workers[@replacement[:old]]
|
|
298
|
+
fresh = @workers[@replacement[:new]]
|
|
299
|
+
if !fresh || %w[failed stopped killed].include?(fresh.state) ||
|
|
300
|
+
(!@replacement[:retired] && now - @replacement[:started] > @config.worker_boot_timeout)
|
|
301
|
+
kill(fresh) if fresh
|
|
302
|
+
@logger.warn("Replacement failed; keeping previous workers")
|
|
303
|
+
@replacement = nil
|
|
304
|
+
@replacement_queue.clear
|
|
305
|
+
@recycle_retry_at = now + @config.worker_boot_timeout
|
|
306
|
+
return true
|
|
307
|
+
end
|
|
308
|
+
if !@replacement[:retired] && fresh.state == "ready" && fresh.stats[:healthy] != false
|
|
309
|
+
retire(old) if old
|
|
310
|
+
reason = @replacement[:reason]
|
|
311
|
+
record_restart(reason)
|
|
312
|
+
@replacement[:retired] = true
|
|
313
|
+
end
|
|
314
|
+
return if old || !@replacement[:retired]
|
|
315
|
+
|
|
316
|
+
@replacement = nil
|
|
317
|
+
end
|
|
318
|
+
while (entry = @replacement_queue.shift)
|
|
319
|
+
old_pid, reason = entry
|
|
320
|
+
old = @workers[old_pid]
|
|
321
|
+
next unless old && !old.term_at
|
|
322
|
+
|
|
323
|
+
fresh = spawn_worker(old.index)
|
|
324
|
+
@replacement = { old: old_pid, new: fresh.pid, reason: reason, started: now, retired: false }
|
|
325
|
+
break
|
|
326
|
+
end
|
|
327
|
+
end
|
|
328
|
+
|
|
329
|
+
def check_recycle
|
|
330
|
+
return if @shutdown_at || @replacement || !@replacement_queue.empty? || @config.worker_recycle.empty?
|
|
331
|
+
return if @recycle_retry_at && now < @recycle_retry_at
|
|
332
|
+
return if @config.bind.end_with?(":0")
|
|
333
|
+
|
|
334
|
+
@workers.each_value do |handle|
|
|
335
|
+
next unless handle.state == "ready" && !handle.term_at
|
|
336
|
+
|
|
337
|
+
@config.worker_recycle.each do |name, limit|
|
|
338
|
+
value = case name
|
|
339
|
+
when :max_requests then handle.stats[:requests_total]
|
|
340
|
+
when :max_rss_mb then handle.stats[:rss_bytes]&./(1024.0 * 1024)
|
|
341
|
+
when :max_pss_mb then handle.stats[:pss_bytes]&./(1024.0 * 1024)
|
|
342
|
+
when :max_lifetime then now - handle.born_at
|
|
343
|
+
end
|
|
344
|
+
next unless value && value >= limit * handle.recycle_factor
|
|
345
|
+
|
|
346
|
+
@replacement_queue << [handle.pid, name.to_s]
|
|
347
|
+
return true
|
|
348
|
+
end
|
|
349
|
+
end
|
|
350
|
+
end
|
|
351
|
+
|
|
352
|
+
def sample_memory
|
|
353
|
+
return unless RUBY_PLATFORM.include?("linux")
|
|
354
|
+
return if @next_memory_at && now < @next_memory_at
|
|
355
|
+
|
|
356
|
+
@next_memory_at = now + @config.status_interval
|
|
357
|
+
@workers.each_value do |handle|
|
|
358
|
+
data = File.read("/proc/#{handle.pid}/smaps_rollup")
|
|
359
|
+
handle.stats[:rss_bytes] = data[/^Rss:\s+(\d+)/, 1].to_i * 1024
|
|
360
|
+
handle.stats[:pss_bytes] = data[/^Pss:\s+(\d+)/, 1].to_i * 1024
|
|
361
|
+
rescue Errno::ENOENT, Errno::ESRCH, Errno::EACCES
|
|
362
|
+
next
|
|
363
|
+
end
|
|
364
|
+
end
|
|
365
|
+
|
|
366
|
+
def check_timeouts
|
|
367
|
+
timestamp = now
|
|
368
|
+
@workers.each_value do |handle|
|
|
369
|
+
next if handle.state == "killed"
|
|
370
|
+
|
|
371
|
+
if handle.term_at && timestamp >= handle.term_at && !handle.kill_at
|
|
372
|
+
handle.state = "draining"
|
|
373
|
+
send_signal("TERM", handle.pid)
|
|
374
|
+
handle.kill_at = timestamp + @config.shutdown_timeout
|
|
375
|
+
end
|
|
376
|
+
if handle.kill_at
|
|
377
|
+
kill(handle) if timestamp >= handle.kill_at
|
|
378
|
+
elsif !@shutdown_at && !handle.term_at
|
|
379
|
+
# An unconsumed heartbeat is not evidence of a stuck worker.
|
|
380
|
+
next if handle.state != "booting" && !@forwarded.empty?
|
|
381
|
+
|
|
382
|
+
elapsed = timestamp - (handle.state == "booting" ? handle.born_at : handle.last_seen)
|
|
383
|
+
timeout = handle.state == "booting" ? @config.worker_boot_timeout : @config.worker_timeout
|
|
384
|
+
if elapsed > timeout
|
|
385
|
+
@logger.error("Worker #{handle.pid} #{handle.state} timeout after #{elapsed.round(2)}s")
|
|
386
|
+
handle.restart_reason = handle.state == "booting" ? "worker_boot_timeout" : "worker_timeout"
|
|
387
|
+
kill(handle)
|
|
388
|
+
end
|
|
389
|
+
end
|
|
390
|
+
end
|
|
391
|
+
end
|
|
392
|
+
|
|
393
|
+
def kill(handle)
|
|
394
|
+
handle.state = "killed"
|
|
395
|
+
send_signal("KILL", handle.pid)
|
|
396
|
+
end
|
|
397
|
+
|
|
398
|
+
def record_restart(reason)
|
|
399
|
+
@metrics.record_restart(reason: reason)
|
|
400
|
+
@forwarded << { type: "restart", pid: Process.pid, reason: reason } if @owner_channel
|
|
401
|
+
end
|
|
402
|
+
|
|
403
|
+
def send_signal(signal, pid)
|
|
404
|
+
Process.kill(signal, pid)
|
|
405
|
+
rescue Errno::ESRCH
|
|
406
|
+
nil
|
|
407
|
+
end
|
|
408
|
+
|
|
409
|
+
def cleanup
|
|
410
|
+
@workers.each_value { |handle| kill(handle) }
|
|
411
|
+
@workers.each_value do |handle| # rubocop:disable Style/CombinableLoops -- kill every child before waiting for any child
|
|
412
|
+
Process.waitpid(handle.pid)
|
|
413
|
+
rescue Errno::ECHILD
|
|
414
|
+
nil
|
|
415
|
+
ensure
|
|
416
|
+
handle.close
|
|
417
|
+
end
|
|
418
|
+
@workers.clear
|
|
419
|
+
@signals&.close
|
|
420
|
+
@admin&.close
|
|
421
|
+
flush_forwarded
|
|
422
|
+
@reports&.close
|
|
423
|
+
end
|
|
424
|
+
end
|
|
425
|
+
end
|
|
426
|
+
end
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Gritz
|
|
4
|
+
module Supervisor
|
|
5
|
+
# Wakes the event loop without doing process or logger work in a signal trap.
|
|
6
|
+
# @api private
|
|
7
|
+
class SignalQueue
|
|
8
|
+
attr_reader :io
|
|
9
|
+
|
|
10
|
+
def initialize(signals:)
|
|
11
|
+
@io, @writer = IO.pipe
|
|
12
|
+
@pending = []
|
|
13
|
+
@previous = {}
|
|
14
|
+
signals.each do |name|
|
|
15
|
+
@previous[name] = Signal.trap(name) do
|
|
16
|
+
@pending << name unless @pending.include?(name)
|
|
17
|
+
@writer.write_nonblock(".", exception: false)
|
|
18
|
+
rescue IOError, Errno::EPIPE
|
|
19
|
+
nil
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
rescue StandardError
|
|
23
|
+
close
|
|
24
|
+
raise
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def drain
|
|
28
|
+
loop do
|
|
29
|
+
break unless @io.read_nonblock(4096, exception: false).is_a?(String)
|
|
30
|
+
end
|
|
31
|
+
# Swap instead of clearing: a signal arriving here belongs to the next drain.
|
|
32
|
+
pending = @pending
|
|
33
|
+
@pending = []
|
|
34
|
+
pending
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def close
|
|
38
|
+
@previous&.each { |name, handler| Signal.trap(name, handler) }
|
|
39
|
+
@previous = {}
|
|
40
|
+
close_in_child
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
def close_in_child
|
|
44
|
+
@io&.close unless @io&.closed?
|
|
45
|
+
@writer&.close unless @writer&.closed?
|
|
46
|
+
end
|
|
47
|
+
end
|
|
48
|
+
end
|
|
49
|
+
end
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
|
|
5
|
+
module Gritz
|
|
6
|
+
module Supervisor
|
|
7
|
+
# Bounded, nonblocking JSON Lines over a worker's dedicated status pipe.
|
|
8
|
+
# @api private
|
|
9
|
+
class StatusChannel
|
|
10
|
+
# ponytail: master snapshots are capped at 64 KiB; chunk reports if clusters outgrow this bound.
|
|
11
|
+
MAX_LINE_BYTES = 64 * 1024
|
|
12
|
+
MAX_READ_BYTES = 64 * 1024
|
|
13
|
+
|
|
14
|
+
attr_reader :io
|
|
15
|
+
|
|
16
|
+
def initialize(io)
|
|
17
|
+
@io = io
|
|
18
|
+
@buffer = +""
|
|
19
|
+
@pending = +""
|
|
20
|
+
@discarding = false
|
|
21
|
+
@closed = io.closed?
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
def write(status)
|
|
25
|
+
return false if closed? || flush_pending.positive?
|
|
26
|
+
|
|
27
|
+
line = "#{JSON.generate(status)}\n"
|
|
28
|
+
return false if line.bytesize > MAX_LINE_BYTES
|
|
29
|
+
|
|
30
|
+
@pending = line
|
|
31
|
+
flush_pending
|
|
32
|
+
true
|
|
33
|
+
rescue JSON::GeneratorError, JSON::NestingError
|
|
34
|
+
false
|
|
35
|
+
rescue IOError, SystemCallError
|
|
36
|
+
close
|
|
37
|
+
false
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
# True means the accepted record is fully delivered; false means retry later.
|
|
41
|
+
def flush
|
|
42
|
+
return false if closed?
|
|
43
|
+
|
|
44
|
+
flush_pending.zero?
|
|
45
|
+
rescue IOError, SystemCallError
|
|
46
|
+
close
|
|
47
|
+
false
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
def read(max_bytes: MAX_READ_BYTES)
|
|
51
|
+
rows = []
|
|
52
|
+
return rows if closed?
|
|
53
|
+
raise ArgumentError, "Invalid status read budget" unless max_bytes.is_a?(Integer) && max_bytes.between?(1, MAX_READ_BYTES)
|
|
54
|
+
|
|
55
|
+
remaining = max_bytes
|
|
56
|
+
while remaining.positive?
|
|
57
|
+
chunk = @io.read_nonblock([4096, remaining].min, exception: false)
|
|
58
|
+
break if chunk == :wait_readable
|
|
59
|
+
|
|
60
|
+
if chunk.nil?
|
|
61
|
+
close
|
|
62
|
+
break
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
remaining -= chunk.bytesize
|
|
66
|
+
consume(chunk, rows)
|
|
67
|
+
end
|
|
68
|
+
rows
|
|
69
|
+
rescue IOError, SystemCallError
|
|
70
|
+
close
|
|
71
|
+
rows
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
def closed? = @closed || @io.closed?
|
|
75
|
+
|
|
76
|
+
def close
|
|
77
|
+
@closed = true
|
|
78
|
+
@buffer.clear
|
|
79
|
+
@pending.clear
|
|
80
|
+
@io.close unless @io.closed?
|
|
81
|
+
nil
|
|
82
|
+
rescue IOError
|
|
83
|
+
nil
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
private
|
|
87
|
+
|
|
88
|
+
def flush_pending
|
|
89
|
+
return 0 if @pending.empty?
|
|
90
|
+
|
|
91
|
+
written = @io.write_nonblock(@pending, exception: false)
|
|
92
|
+
return @pending.bytesize if written == :wait_writable
|
|
93
|
+
|
|
94
|
+
@pending = @pending.byteslice(written..) || +""
|
|
95
|
+
@pending.bytesize
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
def consume(chunk, rows)
|
|
99
|
+
chunk.each_line do |part|
|
|
100
|
+
complete = part.end_with?("\n")
|
|
101
|
+
unless @discarding
|
|
102
|
+
@buffer << part
|
|
103
|
+
if @buffer.bytesize > MAX_LINE_BYTES
|
|
104
|
+
@buffer.clear
|
|
105
|
+
@discarding = true
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
next unless complete
|
|
109
|
+
|
|
110
|
+
unless @discarding
|
|
111
|
+
begin
|
|
112
|
+
row = JSON.parse(@buffer, symbolize_names: true)
|
|
113
|
+
rows << row if row.is_a?(Hash)
|
|
114
|
+
rescue JSON::ParserError
|
|
115
|
+
# A damaged record does not prevent subsequent status updates.
|
|
116
|
+
end
|
|
117
|
+
end
|
|
118
|
+
@buffer.clear
|
|
119
|
+
@discarding = false
|
|
120
|
+
end
|
|
121
|
+
end
|
|
122
|
+
end
|
|
123
|
+
end
|
|
124
|
+
end
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Gritz
|
|
4
|
+
module Supervisor
|
|
5
|
+
# Parent-owned worker state; heartbeat deadlines use the parent's monotonic clock.
|
|
6
|
+
# @api private
|
|
7
|
+
class WorkerHandle
|
|
8
|
+
STATES = %w[booting ready draining failed stopped killed].freeze
|
|
9
|
+
attr_reader :pid, :index, :channel, :born_at, :last_seen, :stats
|
|
10
|
+
attr_accessor :state, :term_at, :kill_at, :recycle_factor, :restart_reason
|
|
11
|
+
|
|
12
|
+
def initialize(pid:, index:, status_io:, now: Process.clock_gettime(Process::CLOCK_MONOTONIC))
|
|
13
|
+
@pid = pid
|
|
14
|
+
@index = index
|
|
15
|
+
@channel = StatusChannel.new(status_io)
|
|
16
|
+
@born_at = @last_seen = now
|
|
17
|
+
@state = "booting"
|
|
18
|
+
@stats = {}
|
|
19
|
+
@recycle_factor = 1.0
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def update(message, now:)
|
|
23
|
+
raise ConfigurationError, "Invalid worker state #{message[:state].inspect}" unless STATES.include?(message[:state])
|
|
24
|
+
|
|
25
|
+
@last_seen = now
|
|
26
|
+
@stats = @stats.slice(:rss_bytes, :pss_bytes).merge(message.except(:pid, :index, :state, :type, :metrics))
|
|
27
|
+
# A late heartbeat must not undo the parent's decision to retire this worker.
|
|
28
|
+
@state = message[:state] unless %w[draining killed].include?(@state)
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
def to_h = @stats.merge(pid: @pid, index: @index, state: @state, retiring: !@term_at.nil?, worker_started_at: @born_at)
|
|
32
|
+
def close = @channel.close
|
|
33
|
+
end
|
|
34
|
+
end
|
|
35
|
+
end
|