gritz-core 0.2.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +13 -0
- data/README.md +4 -0
- data/lib/gritz/channel_registry.rb +42 -0
- data/lib/gritz/cli.rb +31 -5
- data/lib/gritz/client/call_context.rb +60 -0
- data/lib/gritz/client/invocation.rb +84 -0
- data/lib/gritz/client.rb +109 -0
- data/lib/gritz/configuration.rb +23 -9
- data/lib/gritz/context.rb +31 -2
- data/lib/gritz/controller.rb +14 -0
- data/lib/gritz/core/version.rb +1 -1
- data/lib/gritz/dispatcher.rb +9 -3
- data/lib/gritz/errors.rb +3 -1
- data/lib/gritz/metrics/aggregator.rb +119 -0
- data/lib/gritz/metrics/recorder.rb +93 -0
- data/lib/gritz/middleware/exception_mapper.rb +14 -7
- data/lib/gritz/middleware/logging.rb +27 -3
- data/lib/gritz/middleware/metrics.rb +29 -0
- data/lib/gritz/middleware/stack.rb +1 -1
- data/lib/gritz/supervisor/admin_server.rb +121 -0
- data/lib/gritz/supervisor/launcher.rb +381 -0
- data/lib/gritz/supervisor/master.rb +220 -29
- data/lib/gritz/supervisor/status_channel.rb +15 -3
- data/lib/gritz/supervisor/worker_handle.rb +4 -3
- data/lib/gritz/testing/cluster.rb +43 -7
- data/lib/gritz/worker/runner.rb +111 -16
- metadata +30 -1
|
@@ -7,15 +7,19 @@ module Gritz
|
|
|
7
7
|
# Forks workers, monitors their status pipes, and owns every child until reaped.
|
|
8
8
|
# @api public
|
|
9
9
|
class Master
|
|
10
|
-
SIGNALS = %w[TERM INT QUIT TTIN TTOU HUP CHLD].freeze
|
|
10
|
+
SIGNALS = %w[TERM INT QUIT TTIN TTOU HUP CHLD USR1 USR2].freeze
|
|
11
11
|
attr_reader :workers
|
|
12
12
|
|
|
13
|
-
def initialize(config, logger: Logger.new($stdout), status_io: nil)
|
|
13
|
+
def initialize(config, logger: Logger.new($stdout), status_io: nil, owner_channel: nil)
|
|
14
14
|
@config = config
|
|
15
15
|
@logger = logger
|
|
16
16
|
@workers = {}
|
|
17
17
|
@desired = config.workers
|
|
18
|
-
@
|
|
18
|
+
@owner_channel = owner_channel
|
|
19
|
+
@reports = owner_channel || (StatusChannel.new(status_io) if status_io)
|
|
20
|
+
@metrics = Metrics::Aggregator.new
|
|
21
|
+
@forwarded = []
|
|
22
|
+
@replacement_queue = []
|
|
19
23
|
@exit_status = 0
|
|
20
24
|
end
|
|
21
25
|
|
|
@@ -25,6 +29,10 @@ module Gritz
|
|
|
25
29
|
|
|
26
30
|
require "gritz/native"
|
|
27
31
|
@signals = SignalQueue.new(signals: SIGNALS)
|
|
32
|
+
unless @owner_channel
|
|
33
|
+
@admin = AdminServer.new(bind: @config.admin_bind, status: -> { status }, ready: -> { ready? },
|
|
34
|
+
metrics: -> { @metrics.render(workers: @workers.values) }, logger: @logger)
|
|
35
|
+
end
|
|
28
36
|
@guard = ForkGuard.activate(mode: @config.fork_mode == :clean ? @config.fork_guard : :off, logger: @logger)
|
|
29
37
|
@config.preload! if @config.preload_app?
|
|
30
38
|
Process.warmup if @config.preload_app? && Process.respond_to?(:warmup)
|
|
@@ -37,11 +45,25 @@ module Gritz
|
|
|
37
45
|
read_statuses
|
|
38
46
|
reap_children
|
|
39
47
|
check_timeouts
|
|
48
|
+
sample_memory
|
|
49
|
+
check_recycle
|
|
50
|
+
advance_replacement
|
|
40
51
|
maintain_worker_count unless @shutdown_at
|
|
41
|
-
|
|
42
|
-
|
|
52
|
+
flush_forwarded
|
|
53
|
+
@reports&.write(@owner_channel ? status.merge(type: "status") : status) if @forwarded.empty?
|
|
54
|
+
@admin&.poll
|
|
55
|
+
if @owner_channel&.closed?
|
|
56
|
+
@exit_status = 1
|
|
57
|
+
begin_shutdown(immediate: true)
|
|
58
|
+
end
|
|
59
|
+
if @shutdown_at && @workers.empty?
|
|
60
|
+
break if !@owner_channel || @owner_channel.closed? || (@forwarded.empty? && @owner_channel.flush)
|
|
61
|
+
break if now >= @shutdown_at + @config.drain_delay + @config.shutdown_timeout
|
|
62
|
+
end
|
|
43
63
|
|
|
44
|
-
|
|
64
|
+
worker_ios = @forwarded.empty? ? @workers.values.filter_map { |handle| handle.channel.io unless handle.channel.closed? } : []
|
|
65
|
+
owner_ios = !@forwarded.empty? && @owner_channel && !@owner_channel.closed? ? [@owner_channel.io] : nil
|
|
66
|
+
IO.select([@signals.io, *@admin&.ios.to_a, *worker_ios], owner_ios, nil, 0.05)
|
|
45
67
|
end
|
|
46
68
|
@exit_status
|
|
47
69
|
ensure
|
|
@@ -50,7 +72,14 @@ module Gritz
|
|
|
50
72
|
end
|
|
51
73
|
|
|
52
74
|
def status
|
|
53
|
-
{ pid: Process.pid, state: @shutdown_at ? "draining" : "running",
|
|
75
|
+
{ pid: Process.pid, state: @shutdown_at ? "draining" : "running", desired: @desired,
|
|
76
|
+
phased_restart: !@replacement.nil? || !@replacement_queue.empty?, workers: @workers.values.map(&:to_h) }
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
def ready?
|
|
80
|
+
!@shutdown_at && @workers.values.count { |handle|
|
|
81
|
+
handle.state == "ready" && !handle.term_at && handle.stats[:healthy] != false
|
|
82
|
+
} >= @config.min_ready_workers
|
|
54
83
|
end
|
|
55
84
|
|
|
56
85
|
private
|
|
@@ -58,8 +87,9 @@ module Gritz
|
|
|
58
87
|
def now = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
59
88
|
|
|
60
89
|
def maintain_worker_count
|
|
61
|
-
|
|
62
|
-
|
|
90
|
+
active = @workers.values.reject(&:term_at)
|
|
91
|
+
used = active.map(&:index)
|
|
92
|
+
(@desired - active.size).times do
|
|
63
93
|
index = (0...@desired).find { |candidate| !used.include?(candidate) }
|
|
64
94
|
spawn_worker(index)
|
|
65
95
|
used << index
|
|
@@ -82,6 +112,7 @@ module Gritz
|
|
|
82
112
|
# Restore all inherited master traps, including CHLD and resize signals.
|
|
83
113
|
@signals.close
|
|
84
114
|
@reports&.close
|
|
115
|
+
@admin&.close
|
|
85
116
|
@workers.each_value(&:close)
|
|
86
117
|
Transport::Native.postfork_child if experimental
|
|
87
118
|
exit_status = Worker::Runner.new(index: index, status_io: writer, config: @config, logger: @logger).run
|
|
@@ -98,8 +129,11 @@ module Gritz
|
|
|
98
129
|
end
|
|
99
130
|
end
|
|
100
131
|
writer.close
|
|
101
|
-
|
|
132
|
+
handle = WorkerHandle.new(pid: pid, index: index, status_io: reader)
|
|
133
|
+
handle.recycle_factor = 1.0 + (rand * @config.worker_recycle.fetch(:jitter, 0.0))
|
|
134
|
+
@workers[pid] = handle
|
|
102
135
|
@logger.info("Worker #{index} spawned pid=#{pid}")
|
|
136
|
+
handle
|
|
103
137
|
rescue StandardError
|
|
104
138
|
reader&.close unless reader&.closed?
|
|
105
139
|
writer&.close unless writer&.closed?
|
|
@@ -110,36 +144,83 @@ module Gritz
|
|
|
110
144
|
|
|
111
145
|
def read_statuses
|
|
112
146
|
@workers.each_value do |handle|
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
147
|
+
break unless @forwarded.empty?
|
|
148
|
+
|
|
149
|
+
consume_status(handle, handle.channel.read)
|
|
150
|
+
flush_forwarded
|
|
151
|
+
end
|
|
152
|
+
end
|
|
153
|
+
|
|
154
|
+
def consume_status(handle, rows)
|
|
155
|
+
rows.each do |message|
|
|
156
|
+
if message[:type] == "metrics"
|
|
157
|
+
if @metrics.apply(handle, message) && @owner_channel
|
|
158
|
+
@forwarded << message.merge(pid: Process.pid, worker_pid: handle.pid, worker_started_at: handle.born_at)
|
|
118
159
|
end
|
|
119
|
-
|
|
120
|
-
rescue ConfigurationError => e
|
|
121
|
-
@logger.error("Worker #{handle.pid}: #{e.message}")
|
|
122
|
-
kill(handle)
|
|
160
|
+
next
|
|
123
161
|
end
|
|
162
|
+
handle.update(message, now: now)
|
|
163
|
+
if message[:state] == "failed" && !@ever_ready
|
|
164
|
+
@exit_status = 1
|
|
165
|
+
begin_shutdown
|
|
166
|
+
end
|
|
167
|
+
@ever_ready = true if message[:state] == "ready"
|
|
168
|
+
rescue ConfigurationError => e
|
|
169
|
+
@logger.error("Worker #{handle.pid}: #{e.message}")
|
|
170
|
+
kill(handle)
|
|
124
171
|
end
|
|
125
172
|
end
|
|
126
173
|
|
|
174
|
+
def flush_forwarded
|
|
175
|
+
return unless @owner_channel
|
|
176
|
+
|
|
177
|
+
@owner_channel.flush
|
|
178
|
+
@forwarded.shift while !@forwarded.empty? && @owner_channel.write(@forwarded.first)
|
|
179
|
+
end
|
|
180
|
+
|
|
127
181
|
def reap_children
|
|
182
|
+
# Keep a reaped worker's bounded pipe tail queued until owner backpressure clears.
|
|
183
|
+
return unless @forwarded.empty? || @owner_channel&.closed?
|
|
184
|
+
|
|
128
185
|
# Only reap owned children: application hooks can start unrelated subprocesses.
|
|
129
186
|
@workers.each_key do |pid|
|
|
130
187
|
result = Process.waitpid2(pid, Process::WNOHANG)
|
|
131
188
|
next unless result
|
|
132
189
|
|
|
133
|
-
handle = @workers.
|
|
190
|
+
handle = @workers.fetch(pid)
|
|
191
|
+
# A forked application subprocess can retain the writer after the worker exits.
|
|
192
|
+
# Drain the bytes present at reap time without waiting for that subprocess's EOF.
|
|
193
|
+
available = 0
|
|
194
|
+
unless handle.channel.closed?
|
|
195
|
+
size = [0].pack("i")
|
|
196
|
+
# FIONREAD reports queued bytes; Ruby 4.0 removed IO#nread.
|
|
197
|
+
handle.channel.io.ioctl(RUBY_PLATFORM.include?("linux") ? 0x541B : 0x4004667F, size)
|
|
198
|
+
available = size.unpack1("i")
|
|
199
|
+
end
|
|
200
|
+
if available.zero?
|
|
201
|
+
consume_status(handle, handle.channel.read)
|
|
202
|
+
end
|
|
203
|
+
while available.positive?
|
|
204
|
+
bytes = [available, StatusChannel::MAX_READ_BYTES].min
|
|
205
|
+
consume_status(handle, handle.channel.read(max_bytes: bytes))
|
|
206
|
+
available -= bytes
|
|
207
|
+
break if handle.channel.closed?
|
|
208
|
+
end
|
|
209
|
+
@workers.delete(pid)
|
|
210
|
+
@metrics.forget(handle)
|
|
134
211
|
handle.close
|
|
135
212
|
child_status = result.last
|
|
136
213
|
@logger.info("Worker #{handle.index} exited pid=#{pid} status=#{child_status}")
|
|
137
|
-
startup_failed = child_status.exited? && !@ever_ready && !@shutdown_at && !handle.term_at
|
|
138
|
-
shutdown_failed = @shutdown_at &&
|
|
214
|
+
startup_failed = (child_status.exited? || child_status.termsig != Signal.list.fetch("KILL")) && !@ever_ready && !@shutdown_at && !handle.term_at
|
|
215
|
+
shutdown_failed = @shutdown_at && !child_status.success?
|
|
139
216
|
if handle.state != "killed" && (startup_failed || shutdown_failed)
|
|
140
217
|
@exit_status = 1
|
|
141
218
|
begin_shutdown(immediate: true) if startup_failed
|
|
142
219
|
end
|
|
220
|
+
if !@shutdown_at && !handle.term_at && @workers.values.count { |worker| !worker.term_at } < @desired
|
|
221
|
+
record_restart(handle.restart_reason || "worker_exit")
|
|
222
|
+
end
|
|
223
|
+
break unless @forwarded.empty?
|
|
143
224
|
rescue Errno::ECHILD
|
|
144
225
|
@workers.delete(pid)&.close
|
|
145
226
|
end
|
|
@@ -150,37 +231,136 @@ module Gritz
|
|
|
150
231
|
when "TERM", "INT" then begin_shutdown
|
|
151
232
|
when "QUIT" then begin_shutdown(immediate: true)
|
|
152
233
|
when "TTIN"
|
|
153
|
-
if
|
|
234
|
+
if @replacement || !@replacement_queue.empty?
|
|
235
|
+
@logger.warn("Wait for phased restart before resizing workers")
|
|
236
|
+
elsif !@shutdown_at && @config.bind.end_with?(":0")
|
|
154
237
|
@logger.warn("Cannot add a reuseport worker with port 0; configure a fixed bind port")
|
|
155
238
|
elsif !@shutdown_at
|
|
156
239
|
@desired += 1
|
|
157
240
|
end
|
|
158
241
|
when "TTOU"
|
|
159
|
-
if
|
|
242
|
+
if @replacement || !@replacement_queue.empty?
|
|
243
|
+
@logger.warn("Wait for phased restart before resizing workers")
|
|
244
|
+
elsif !@shutdown_at && @desired > 1
|
|
160
245
|
@desired -= 1
|
|
161
|
-
retire(@workers.values.reject(&:term_at).max_by(&:index))
|
|
246
|
+
retire(@workers.values.reject(&:term_at).max_by(&:index), delay: 0)
|
|
162
247
|
end
|
|
163
248
|
when "HUP"
|
|
164
249
|
@logger.reopen
|
|
165
250
|
@workers.each_key { |pid| send_signal("HUP", pid) }
|
|
166
251
|
@logger.info("Log reopened")
|
|
252
|
+
when "USR1"
|
|
253
|
+
if fixed_port? && !@shutdown_at && !@replacement && @replacement_queue.empty?
|
|
254
|
+
@replacement_queue = @workers.values.reject(&:term_at).sort_by(&:index).map { |handle| [handle.pid, "phased_restart"] }
|
|
255
|
+
end
|
|
256
|
+
when "USR2"
|
|
257
|
+
if fixed_port? && !@shutdown_at
|
|
258
|
+
if @owner_channel
|
|
259
|
+
@forwarded << { type: "reexec", pid: Process.pid } unless @forwarded.any? { |row| row[:type] == "reexec" }
|
|
260
|
+
else
|
|
261
|
+
@logger.warn("USR2 requires the gritz launcher")
|
|
262
|
+
end
|
|
263
|
+
end
|
|
167
264
|
end
|
|
168
265
|
end
|
|
169
266
|
|
|
170
267
|
def begin_shutdown(immediate: false)
|
|
171
268
|
@shutdown_at ||= now
|
|
269
|
+
@replacement_queue.clear
|
|
270
|
+
@replacement = nil
|
|
172
271
|
if immediate
|
|
173
272
|
@workers.each_value { |handle| kill(handle) }
|
|
174
273
|
else
|
|
175
|
-
@workers.each_value { |handle| handle
|
|
274
|
+
@workers.each_value { |handle| retire(handle) unless handle.term_at }
|
|
176
275
|
end
|
|
177
276
|
end
|
|
178
277
|
|
|
179
|
-
def retire(handle)
|
|
278
|
+
def retire(handle, delay: @config.drain_delay)
|
|
180
279
|
return unless handle
|
|
181
280
|
|
|
182
|
-
handle.term_at = now
|
|
281
|
+
handle.term_at = now + delay
|
|
183
282
|
handle.state = "draining"
|
|
283
|
+
send_signal("USR1", handle.pid)
|
|
284
|
+
end
|
|
285
|
+
|
|
286
|
+
def fixed_port?
|
|
287
|
+
return true unless @config.bind.end_with?(":0")
|
|
288
|
+
|
|
289
|
+
@logger.warn("Worker replacement requires a fixed bind port")
|
|
290
|
+
false
|
|
291
|
+
end
|
|
292
|
+
|
|
293
|
+
def advance_replacement
|
|
294
|
+
return if @shutdown_at || !@forwarded.empty?
|
|
295
|
+
|
|
296
|
+
if @replacement
|
|
297
|
+
old = @workers[@replacement[:old]]
|
|
298
|
+
fresh = @workers[@replacement[:new]]
|
|
299
|
+
if !fresh || %w[failed stopped killed].include?(fresh.state) ||
|
|
300
|
+
(!@replacement[:retired] && now - @replacement[:started] > @config.worker_boot_timeout)
|
|
301
|
+
kill(fresh) if fresh
|
|
302
|
+
@logger.warn("Replacement failed; keeping previous workers")
|
|
303
|
+
@replacement = nil
|
|
304
|
+
@replacement_queue.clear
|
|
305
|
+
@recycle_retry_at = now + @config.worker_boot_timeout
|
|
306
|
+
return true
|
|
307
|
+
end
|
|
308
|
+
if !@replacement[:retired] && fresh.state == "ready" && fresh.stats[:healthy] != false
|
|
309
|
+
retire(old) if old
|
|
310
|
+
reason = @replacement[:reason]
|
|
311
|
+
record_restart(reason)
|
|
312
|
+
@replacement[:retired] = true
|
|
313
|
+
end
|
|
314
|
+
return if old || !@replacement[:retired]
|
|
315
|
+
|
|
316
|
+
@replacement = nil
|
|
317
|
+
end
|
|
318
|
+
while (entry = @replacement_queue.shift)
|
|
319
|
+
old_pid, reason = entry
|
|
320
|
+
old = @workers[old_pid]
|
|
321
|
+
next unless old && !old.term_at
|
|
322
|
+
|
|
323
|
+
fresh = spawn_worker(old.index)
|
|
324
|
+
@replacement = { old: old_pid, new: fresh.pid, reason: reason, started: now, retired: false }
|
|
325
|
+
break
|
|
326
|
+
end
|
|
327
|
+
end
|
|
328
|
+
|
|
329
|
+
def check_recycle
|
|
330
|
+
return if @shutdown_at || @replacement || !@replacement_queue.empty? || @config.worker_recycle.empty?
|
|
331
|
+
return if @recycle_retry_at && now < @recycle_retry_at
|
|
332
|
+
return if @config.bind.end_with?(":0")
|
|
333
|
+
|
|
334
|
+
@workers.each_value do |handle|
|
|
335
|
+
next unless handle.state == "ready" && !handle.term_at
|
|
336
|
+
|
|
337
|
+
@config.worker_recycle.each do |name, limit|
|
|
338
|
+
value = case name
|
|
339
|
+
when :max_requests then handle.stats[:requests_total]
|
|
340
|
+
when :max_rss_mb then handle.stats[:rss_bytes]&./(1024.0 * 1024)
|
|
341
|
+
when :max_pss_mb then handle.stats[:pss_bytes]&./(1024.0 * 1024)
|
|
342
|
+
when :max_lifetime then now - handle.born_at
|
|
343
|
+
end
|
|
344
|
+
next unless value && value >= limit * handle.recycle_factor
|
|
345
|
+
|
|
346
|
+
@replacement_queue << [handle.pid, name.to_s]
|
|
347
|
+
return true
|
|
348
|
+
end
|
|
349
|
+
end
|
|
350
|
+
end
|
|
351
|
+
|
|
352
|
+
def sample_memory
|
|
353
|
+
return unless RUBY_PLATFORM.include?("linux")
|
|
354
|
+
return if @next_memory_at && now < @next_memory_at
|
|
355
|
+
|
|
356
|
+
@next_memory_at = now + @config.status_interval
|
|
357
|
+
@workers.each_value do |handle|
|
|
358
|
+
data = File.read("/proc/#{handle.pid}/smaps_rollup")
|
|
359
|
+
handle.stats[:rss_bytes] = data[/^Rss:\s+(\d+)/, 1].to_i * 1024
|
|
360
|
+
handle.stats[:pss_bytes] = data[/^Pss:\s+(\d+)/, 1].to_i * 1024
|
|
361
|
+
rescue Errno::ENOENT, Errno::ESRCH, Errno::EACCES
|
|
362
|
+
next
|
|
363
|
+
end
|
|
184
364
|
end
|
|
185
365
|
|
|
186
366
|
def check_timeouts
|
|
@@ -195,11 +375,15 @@ module Gritz
|
|
|
195
375
|
end
|
|
196
376
|
if handle.kill_at
|
|
197
377
|
kill(handle) if timestamp >= handle.kill_at
|
|
198
|
-
elsif !@shutdown_at
|
|
378
|
+
elsif !@shutdown_at && !handle.term_at
|
|
379
|
+
# An unconsumed heartbeat is not evidence of a stuck worker.
|
|
380
|
+
next if handle.state != "booting" && !@forwarded.empty?
|
|
381
|
+
|
|
199
382
|
elapsed = timestamp - (handle.state == "booting" ? handle.born_at : handle.last_seen)
|
|
200
383
|
timeout = handle.state == "booting" ? @config.worker_boot_timeout : @config.worker_timeout
|
|
201
384
|
if elapsed > timeout
|
|
202
385
|
@logger.error("Worker #{handle.pid} #{handle.state} timeout after #{elapsed.round(2)}s")
|
|
386
|
+
handle.restart_reason = handle.state == "booting" ? "worker_boot_timeout" : "worker_timeout"
|
|
203
387
|
kill(handle)
|
|
204
388
|
end
|
|
205
389
|
end
|
|
@@ -211,6 +395,11 @@ module Gritz
|
|
|
211
395
|
send_signal("KILL", handle.pid)
|
|
212
396
|
end
|
|
213
397
|
|
|
398
|
+
def record_restart(reason)
|
|
399
|
+
@metrics.record_restart(reason: reason)
|
|
400
|
+
@forwarded << { type: "restart", pid: Process.pid, reason: reason } if @owner_channel
|
|
401
|
+
end
|
|
402
|
+
|
|
214
403
|
def send_signal(signal, pid)
|
|
215
404
|
Process.kill(signal, pid)
|
|
216
405
|
rescue Errno::ESRCH
|
|
@@ -228,6 +417,8 @@ module Gritz
|
|
|
228
417
|
end
|
|
229
418
|
@workers.clear
|
|
230
419
|
@signals&.close
|
|
420
|
+
@admin&.close
|
|
421
|
+
flush_forwarded
|
|
231
422
|
@reports&.close
|
|
232
423
|
end
|
|
233
424
|
end
|
|
@@ -28,7 +28,8 @@ module Gritz
|
|
|
28
28
|
return false if line.bytesize > MAX_LINE_BYTES
|
|
29
29
|
|
|
30
30
|
@pending = line
|
|
31
|
-
flush_pending
|
|
31
|
+
flush_pending
|
|
32
|
+
true
|
|
32
33
|
rescue JSON::GeneratorError, JSON::NestingError
|
|
33
34
|
false
|
|
34
35
|
rescue IOError, SystemCallError
|
|
@@ -36,11 +37,22 @@ module Gritz
|
|
|
36
37
|
false
|
|
37
38
|
end
|
|
38
39
|
|
|
39
|
-
|
|
40
|
+
# True means the accepted record is fully delivered; false means retry later.
|
|
41
|
+
def flush
|
|
42
|
+
return false if closed?
|
|
43
|
+
|
|
44
|
+
flush_pending.zero?
|
|
45
|
+
rescue IOError, SystemCallError
|
|
46
|
+
close
|
|
47
|
+
false
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
def read(max_bytes: MAX_READ_BYTES)
|
|
40
51
|
rows = []
|
|
41
52
|
return rows if closed?
|
|
53
|
+
raise ArgumentError, "Invalid status read budget" unless max_bytes.is_a?(Integer) && max_bytes.between?(1, MAX_READ_BYTES)
|
|
42
54
|
|
|
43
|
-
remaining =
|
|
55
|
+
remaining = max_bytes
|
|
44
56
|
while remaining.positive?
|
|
45
57
|
chunk = @io.read_nonblock([4096, remaining].min, exception: false)
|
|
46
58
|
break if chunk == :wait_readable
|
|
@@ -7,7 +7,7 @@ module Gritz
|
|
|
7
7
|
class WorkerHandle
|
|
8
8
|
STATES = %w[booting ready draining failed stopped killed].freeze
|
|
9
9
|
attr_reader :pid, :index, :channel, :born_at, :last_seen, :stats
|
|
10
|
-
attr_accessor :state, :term_at, :kill_at
|
|
10
|
+
attr_accessor :state, :term_at, :kill_at, :recycle_factor, :restart_reason
|
|
11
11
|
|
|
12
12
|
def initialize(pid:, index:, status_io:, now: Process.clock_gettime(Process::CLOCK_MONOTONIC))
|
|
13
13
|
@pid = pid
|
|
@@ -16,18 +16,19 @@ module Gritz
|
|
|
16
16
|
@born_at = @last_seen = now
|
|
17
17
|
@state = "booting"
|
|
18
18
|
@stats = {}
|
|
19
|
+
@recycle_factor = 1.0
|
|
19
20
|
end
|
|
20
21
|
|
|
21
22
|
def update(message, now:)
|
|
22
23
|
raise ConfigurationError, "Invalid worker state #{message[:state].inspect}" unless STATES.include?(message[:state])
|
|
23
24
|
|
|
24
25
|
@last_seen = now
|
|
25
|
-
@stats = message.except(:pid, :index, :state)
|
|
26
|
+
@stats = @stats.slice(:rss_bytes, :pss_bytes).merge(message.except(:pid, :index, :state, :type, :metrics))
|
|
26
27
|
# A late heartbeat must not undo the parent's decision to retire this worker.
|
|
27
28
|
@state = message[:state] unless %w[draining killed].include?(@state)
|
|
28
29
|
end
|
|
29
30
|
|
|
30
|
-
def to_h = @stats.merge(pid: @pid, index: @index, state: @state)
|
|
31
|
+
def to_h = @stats.merge(pid: @pid, index: @index, state: @state, retiring: !@term_at.nil?, worker_started_at: @born_at)
|
|
31
32
|
def close = @channel.close
|
|
32
33
|
end
|
|
33
34
|
end
|
|
@@ -35,12 +35,13 @@ module Gritz
|
|
|
35
35
|
def start
|
|
36
36
|
raise ArgumentError, "cluster is already started" if @pid
|
|
37
37
|
|
|
38
|
+
Supervisor::Launcher.enable_subreaper!
|
|
38
39
|
@reader, writer = IO.pipe
|
|
39
40
|
@channel = Supervisor::StatusChannel.new(@reader)
|
|
40
41
|
@log = Tempfile.new(["gritz-cluster", ".log"])
|
|
41
42
|
command = @command || [
|
|
42
43
|
RbConfig.ruby, "-I", $LOAD_PATH.join(File::PATH_SEPARATOR), "-rgritz/core", "-e",
|
|
43
|
-
"exit Gritz::CLI.new(status_io: IO.for_fd(3)).run(ARGV)", "--", "start", "-C", @config_path
|
|
44
|
+
"exit Gritz::CLI.new(status_io: IO.for_fd(3), launch: true).run(ARGV)", "--", "start", "-C", @config_path
|
|
44
45
|
]
|
|
45
46
|
@pid = Process.spawn(@env, *command, 3 => writer, out: @log, err: @log, pgroup: true)
|
|
46
47
|
self
|
|
@@ -61,6 +62,8 @@ module Gritz
|
|
|
61
62
|
|
|
62
63
|
def workers = status.fetch(:workers, [])
|
|
63
64
|
|
|
65
|
+
def master_pid = status[:pid] || @pid
|
|
66
|
+
|
|
64
67
|
def logs
|
|
65
68
|
return @logs || "" unless @log && !@log.closed?
|
|
66
69
|
|
|
@@ -83,12 +86,13 @@ module Gritz
|
|
|
83
86
|
deadline = monotonic + timeout
|
|
84
87
|
loop do
|
|
85
88
|
snapshot = status
|
|
89
|
+
active_workers = snapshot.fetch(:workers, []).reject { |worker| worker[:retiring] }
|
|
86
90
|
ready = if state == "ready"
|
|
87
|
-
snapshot[:state] == "running" &&
|
|
91
|
+
snapshot[:state] == "running" && active_workers.any? && active_workers.all? { |worker| worker[:state] == "ready" }
|
|
88
92
|
else
|
|
89
93
|
state.nil? || snapshot[:state] == state
|
|
90
94
|
end
|
|
91
|
-
count = workers.nil? ||
|
|
95
|
+
count = workers.nil? || active_workers.size == workers
|
|
92
96
|
return self if ready && count && (!block_given? || yield(snapshot))
|
|
93
97
|
|
|
94
98
|
if exited?
|
|
@@ -120,14 +124,16 @@ module Gritz
|
|
|
120
124
|
wait(timeout:)
|
|
121
125
|
self
|
|
122
126
|
rescue Timeout::Error
|
|
127
|
+
signal("QUIT") if status[:owner_pid] && !exited?
|
|
123
128
|
begin
|
|
124
|
-
|
|
125
|
-
rescue
|
|
126
|
-
|
|
129
|
+
wait(timeout: 0.5)
|
|
130
|
+
rescue Timeout::Error
|
|
131
|
+
owned_groups.each { |pid| kill_group(pid) }
|
|
132
|
+
wait(timeout: 5)
|
|
127
133
|
end
|
|
128
|
-
wait(timeout: 5)
|
|
129
134
|
self
|
|
130
135
|
ensure
|
|
136
|
+
cleanup_groups if @pid && @exit_status
|
|
131
137
|
@channel&.close
|
|
132
138
|
@logs = logs
|
|
133
139
|
@log&.close!
|
|
@@ -137,6 +143,36 @@ module Gritz
|
|
|
137
143
|
|
|
138
144
|
def monotonic = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
139
145
|
|
|
146
|
+
def owned_groups
|
|
147
|
+
[@pid, *status.fetch(:masters, []).map { |master| master[:pid] }].compact.uniq
|
|
148
|
+
end
|
|
149
|
+
|
|
150
|
+
def kill_group(pid)
|
|
151
|
+
Process.kill("KILL", -pid)
|
|
152
|
+
rescue Errno::ESRCH
|
|
153
|
+
nil
|
|
154
|
+
rescue Errno::EPERM
|
|
155
|
+
# Darwin may report EPERM for an already-disappeared process group.
|
|
156
|
+
begin
|
|
157
|
+
Process.kill(0, pid)
|
|
158
|
+
rescue Errno::ESRCH
|
|
159
|
+
return
|
|
160
|
+
end
|
|
161
|
+
raise
|
|
162
|
+
end
|
|
163
|
+
|
|
164
|
+
def cleanup_groups
|
|
165
|
+
groups = owned_groups.reject { |pid| pid == @pid }
|
|
166
|
+
groups.each { |pid| kill_group(pid) }
|
|
167
|
+
# Terminate every group before waiting for any adopted children.
|
|
168
|
+
groups.each do |pid| # rubocop:disable Style/CombinableLoops
|
|
169
|
+
loop { Process.waitpid(-pid) }
|
|
170
|
+
rescue Errno::ECHILD
|
|
171
|
+
nil
|
|
172
|
+
end
|
|
173
|
+
@status = @status.merge(masters: [])
|
|
174
|
+
end
|
|
175
|
+
|
|
140
176
|
def exited?
|
|
141
177
|
return true if @exit_status
|
|
142
178
|
return false unless @pid
|