gritz-core 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 9d69bbe23da7b5ec3c973bdbe3a2c2c1d1596b6632695f446ae6fb242d0d1f94
4
- data.tar.gz: 89929d525c1709e164057aa7d684317db596734395371cf942019832cf7189bd
3
+ metadata.gz: 023ead58f761660de79d7168434d85be2f92e583073fab043b18b18fd14035c5
4
+ data.tar.gz: f319a9c5f6a190fad09e78667072ee22a5994dfe1f5c2e4f12287f3213865e1a
5
5
  SHA512:
6
- metadata.gz: c4c61c0a88eedb466c3f556f9cd9861468f496b0ba6303be58133d58c6761d8d6fa6a1a9c4b51e7dd6bec168b8207aa8b82562d096f0ead5f8745e351a67c790
7
- data.tar.gz: 5e7d9a717648f45c0f3774f3de35511608d6b6fa7fc5fc6f6be61fbf4d7b874c1cfafdd348b9441e0f37a528a30d341f3eddfa45aa50760255a42f7b644a58cf
6
+ metadata.gz: d531f928a935dd66d9a3459919fff84263309e6821da90332a841170b1bc268e2ccc711394a5873590611883136d44987d0d059794520d1335d788ad37815660
7
+ data.tar.gz: '08b50b33dcbb22fdde34c4052f2c15e82cfb1562bc83f450966f98879e4f0d8bc64cf50f8bb3517fae12ebfe4ac4841928f5ab5f5032a96eabd006711f8e44dc'
data/CHANGELOG.md CHANGED
@@ -1,5 +1,11 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.2.0
4
+
5
+ - Supervise forked workers with boot and heartbeat timeouts, automatic replacement, graceful shutdown, lifecycle hooks, dynamic worker counts and log reopening.
6
+ - Detect unsafe master-side gRPC initialization with ForkGuard and `gritz check`.
7
+ - Add `Gritz::Testing::Cluster` for process lifecycle integration tests.
8
+
3
9
  ## 0.1.0
4
10
 
5
11
  Initial release.
data/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # Gritz Core
2
2
 
3
- Transport-independent controllers, routing, middleware, configuration and network-free testing for Ruby gRPC applications. Requires CRuby 3.3 or later. This gem has no grpc dependency and never loads a transport by itself.
3
+ Transport-independent controllers, routing, middleware, configuration, forked-worker supervision and network-free testing for Ruby gRPC applications. Requires CRuby 3.3 or later. This gem has no grpc dependency. Requiring the core does not load a transport; starting a server loads the selected adapter.
4
4
 
5
5
  ```ruby
6
6
  require "gritz/core"
data/lib/gritz/cli.rb CHANGED
@@ -4,13 +4,14 @@ require "optparse"
4
4
  require "logger"
5
5
 
6
6
  module Gritz
7
- # Command line entry point for the single-process server and route listing.
7
+ # Starts a server, lists routes, or checks configuration and fork safety.
8
8
  # @api public
9
9
  class CLI
10
- def initialize(stdout: $stdout, stderr: $stderr, env: ENV)
10
+ def initialize(stdout: $stdout, stderr: $stderr, env: ENV, status_io: nil)
11
11
  @stdout = stdout
12
12
  @stderr = stderr
13
13
  @env = env
14
+ @status_io = status_io
14
15
  end
15
16
 
16
17
  def run(argv)
@@ -18,7 +19,7 @@ module Gritz
18
19
  path = nil
19
20
  overrides = {}
20
21
  parser = OptionParser.new do |options|
21
- options.banner = "Usage: gritz [start|routes] [-C config/gritz.rb] [options]"
22
+ options.banner = "Usage: gritz [start|routes|check] [-C config/gritz.rb] [options]"
22
23
  options.on("-C", "--config PATH", "Configuration file") { |value| path = value }
23
24
  options.on("--workers N", Integer) { |value| overrides[:workers] = value }
24
25
  options.on("--threads N", Integer) { |value| overrides[:threads] = value }
@@ -35,52 +36,51 @@ module Gritz
35
36
  end
36
37
  parser.parse!(args)
37
38
  command = args.shift || "start"
38
- raise ConfigurationError, "Unknown command #{command}" unless %w[start routes].include?(command)
39
+ raise ConfigurationError, "Unknown command #{command}" unless %w[start routes check].include?(command)
39
40
  raise ConfigurationError, "Unexpected argument #{args.first}" unless args.empty?
40
41
 
41
42
  path ||= "config/gritz.rb" if File.file?("config/gritz.rb")
43
+ guard = ForkGuard.activate(mode: :record) if command != "routes"
44
+ require "gritz/native" if command != "routes"
42
45
  config = Configuration.load(path: path, env: @env, overrides: overrides)
43
- config.validate_single_process! if command == "start"
44
- config.preload! if config.preload_app?
45
46
  logger = Logger.new(@stdout)
46
47
  logger.formatter = ->(_severity, _time, _progname, message) { "#{message}\n" }
47
- router = Router.new(controllers: config.controllers, strict: config.strict_routes, logger: logger)
48
+ if command == "check"
49
+ config.preload! if config.preload_app?
50
+ guard.violations.each { |violation| @stderr.puts(violation.message) }
51
+ return 1 unless guard.violations.empty?
52
+
53
+ config.validate_runtime!
54
+ Router.new(controllers: config.controllers, strict: config.strict_routes, logger: logger)
55
+ @stdout.puts "Configuration and fork safety checks passed"
56
+ elsif command == "start"
57
+ config.validate_runtime!
58
+ if config.workers.positive? && config.fork_mode == :clean
59
+ raise guard.violations.first if config.fork_guard == :raise && !guard.violations.empty?
60
+
61
+ guard.violations.each { |violation| logger.warn(violation.message) } if config.fork_guard == :warn
62
+ end
63
+ @stdout.sync = true if @stdout.respond_to?(:sync=)
64
+ return Supervisor::Master.new(config, logger: logger, status_io: @status_io).run if config.workers.positive?
65
+
66
+ ForkGuard.deactivate
67
+ config.preload! if config.preload_app?
68
+ return Worker::Runner.new(index: 0, config: config, logger: logger).run
69
+ else
70
+ config.preload! if config.preload_app?
71
+ router = Router.new(controllers: config.controllers, strict: config.strict_routes, logger: logger)
72
+ end
48
73
  if command == "routes"
49
74
  router.routes.each_value do |route|
50
75
  @stdout.puts "#{route.full_name} #{route.kind} #{route.controller}##{route.action}"
51
76
  end
52
- else
53
- start(config, router, logger)
54
77
  end
55
78
  0
56
- rescue ConfigurationError, OptionParser::ParseError, ArgumentError, Errno::ENOENT, SyntaxError, LoadError => e
79
+ rescue ConfigurationError, ForkGuard::Violation, ArgumentError, Errno::ENOENT, SyntaxError, LoadError, RuntimeError => e
57
80
  @stderr.puts "gritz: #{e.message}"
58
81
  1
59
- end
60
-
61
- private
62
-
63
- def start(config, router, logger)
64
- @stdout.sync = true if @stdout.respond_to?(:sync=)
65
- # Load the adapter only when starting; route inspection stays transport-independent.
66
- require "gritz/native"
67
- dispatcher = Dispatcher.new(router: router, middleware: config.middleware, logger: logger)
68
- adapter = Transport::Native.new(config: config, dispatcher: dispatcher, logger: logger)
69
- signals = []
70
- previous = %w[TERM INT QUIT].to_h { |name| [name, Signal.trap(name) { signals << name }] }
71
- config.run_hooks(:on_worker_boot, 0)
72
- port = adapter.bind(config.bind)
73
- adapter.start
74
- logger.info("Gritz #{Core::VERSION} listening on #{config.bind.sub(/:\d+\z/, ":#{port}")}")
75
- while signals.empty?
76
- sleep 0.05
77
- raise ConfigurationError, "gRPC server stopped unexpectedly" unless adapter.running?
78
- end
79
- signals.shift == "QUIT" ? adapter.kill : adapter.stop(deadline: Time.now + config.shutdown_timeout)
80
82
  ensure
81
- previous&.each { |name, handler| Signal.trap(name, handler) }
82
- adapter&.kill if adapter&.running?
83
- config.run_hooks(:on_worker_shutdown, 0)
83
+ ForkGuard.deactivate if guard
84
84
  end
85
85
  end
86
86
  end
@@ -3,7 +3,7 @@
3
3
  require "json"
4
4
 
5
5
  module Gritz
6
- # Startup settings. Process-related settings are reserved for the supervisor.
6
+ # Validated startup settings for single-process and supervised servers.
7
7
  # @api public
8
8
  class Configuration
9
9
  DEFAULTS = {
@@ -46,6 +46,7 @@ module Gritz
46
46
  DSL.new(config).evaluate(path) if path
47
47
  env.each do |key, value|
48
48
  next unless key.start_with?("GRITZ_")
49
+ next if key == "GRITZ_RELEASE" # Dependency selection in the release workflow, not a runtime setting.
49
50
 
50
51
  name = key.delete_prefix("GRITZ_").downcase.to_sym
51
52
  raise ConfigurationError, "Unknown environment setting #{key}" unless DEFAULTS.key?(name)
@@ -104,22 +105,28 @@ module Gritz
104
105
  def preload_app? = preload_app
105
106
 
106
107
  # Fail before allocating transport resources for unsupported release features.
107
- def validate_single_process!
108
+ def validate_runtime!
108
109
  validate!
109
- raise ConfigurationError, "v0.1 supports workers 0; the supervisor is planned for v0.2" unless workers.zero?
110
- raise ConfigurationError, "v0.1 supports transport :native" unless transport == :native
111
- raise ConfigurationError, "v0.1 supports listener_strategy :reuseport" unless listener_strategy == :reuseport
110
+ raise ConfigurationError, "This release supports transport :native" unless transport == :native
111
+ raise ConfigurationError, "This release supports listener_strategy :reuseport" unless listener_strategy == :reuseport
112
112
  raise ConfigurationError, "TLS is planned for v0.3" unless tls.empty?
113
- raise ConfigurationError, "worker_recycle requires the supervisor" unless worker_recycle.empty?
113
+ raise ConfigurationError, "worker_recycle is planned for v0.3" unless worker_recycle.empty?
114
114
  raise ConfigurationError, "Health checks are planned for v0.3" unless health_checks.empty?
115
- raise ConfigurationError, "v0.1 supports log_format :json" unless log_format == :json
116
- raise ConfigurationError, "v0.1 reserves metrics_backend :pipe; metrics export is planned for v0.3" unless metrics_backend == :pipe
117
- raise ConfigurationError, "grpc_fork_support requires the supervisor" unless fork_mode == :clean
115
+ raise ConfigurationError, "This release supports log_format :json" unless log_format == :json
116
+ raise ConfigurationError, "This release reserves metrics_backend :pipe; metrics export is planned for v0.3" unless metrics_backend == :pipe
117
+ raise ConfigurationError, "grpc_fork_support requires workers > 0" if workers.zero? && fork_mode != :clean
118
118
  raise ConfigurationError, "at least one controller must be registered" if controllers.empty?
119
119
 
120
120
  self
121
121
  end
122
122
 
123
+ def validate_single_process!
124
+ validate_runtime!
125
+ raise ConfigurationError, "Testing::Server requires workers 0; use Testing::Cluster for supervised servers" unless workers.zero?
126
+
127
+ self
128
+ end
129
+
123
130
  def add_preloader(&block)
124
131
  @preload_app = true
125
132
  @preloaders << block if block
@@ -2,6 +2,6 @@
2
2
 
3
3
  module Gritz
4
4
  module Core
5
- VERSION = "0.1.0"
5
+ VERSION = "0.2.0"
6
6
  end
7
7
  end
@@ -0,0 +1,75 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Gritz
4
+ # Detects transport resource constructors in the master before workers fork.
5
+ # @api public
6
+ class ForkGuard
7
+ class Violation < StandardError
8
+ attr_reader :klass, :locations
9
+
10
+ def initialize(klass, locations)
11
+ @klass = klass
12
+ @locations = locations
13
+ super("gRPC object was created in the master process before fork.\n " \
14
+ "#{klass}.new\n#{locations.map { |location| " from #{location}" }.join("\n")}\n" \
15
+ "Fix: create it in an on_worker_boot hook.")
16
+ end
17
+ end
18
+
19
+ module ConstructorHook
20
+ def new(...)
21
+ Gritz::ForkGuard.check!(self)
22
+ super
23
+ end
24
+ end
25
+
26
+ attr_reader :mode, :logger, :master_pid, :violations
27
+
28
+ class << self
29
+ attr_reader :current
30
+
31
+ def activate(mode: :raise, logger: nil, master_pid: Process.pid)
32
+ @current = new(mode:, logger:, master_pid:)
33
+ end
34
+
35
+ def deactivate
36
+ @current = nil
37
+ end
38
+
39
+ def install(klass)
40
+ klass.singleton_class.prepend(ConstructorHook) unless klass.singleton_class.ancestors.include?(ConstructorHook)
41
+ end
42
+
43
+ def master?
44
+ @current&.master? || false
45
+ end
46
+
47
+ def check!(klass)
48
+ guard = @current
49
+ return unless guard && guard.mode != :off && guard.master?
50
+
51
+ locations = caller_locations(1, 8).reject { |location| location.path == __FILE__ }
52
+ violation = Violation.new(klass, locations)
53
+ guard.violations << violation
54
+ case guard.mode
55
+ when :raise then raise violation
56
+ when :warn
57
+ guard.logger ? guard.logger.warn(violation.message) : Kernel.warn(violation.message)
58
+ end
59
+ end
60
+ end
61
+
62
+ def initialize(mode:, logger:, master_pid:)
63
+ raise ArgumentError, "fork guard mode must be :raise, :warn, :off, or :record" unless %i[raise warn off record].include?(mode)
64
+
65
+ @mode = mode
66
+ @logger = logger
67
+ @master_pid = master_pid
68
+ @violations = []
69
+ end
70
+
71
+ def master?
72
+ Process.pid == @master_pid
73
+ end
74
+ end
75
+ end
@@ -0,0 +1,235 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "logger"
4
+
5
+ module Gritz
6
+ module Supervisor
7
+ # Forks workers, monitors their status pipes, and owns every child until reaped.
8
+ # @api public
9
+ class Master
10
+ SIGNALS = %w[TERM INT QUIT TTIN TTOU HUP CHLD].freeze
11
+ attr_reader :workers
12
+
13
+ def initialize(config, logger: Logger.new($stdout), status_io: nil)
14
+ @config = config
15
+ @logger = logger
16
+ @workers = {}
17
+ @desired = config.workers
18
+ @reports = StatusChannel.new(status_io) if status_io
19
+ @exit_status = 0
20
+ end
21
+
22
+ def run
23
+ @config.validate_runtime!
24
+ raise ConfigurationError, "Supervisor requires workers > 0" unless @desired.positive?
25
+
26
+ require "gritz/native"
27
+ @signals = SignalQueue.new(signals: SIGNALS)
28
+ @guard = ForkGuard.activate(mode: @config.fork_mode == :clean ? @config.fork_guard : :off, logger: @logger)
29
+ @config.preload! if @config.preload_app?
30
+ Process.warmup if @config.preload_app? && Process.respond_to?(:warmup)
31
+ if @config.workers > 1 && !RUBY_PLATFORM.include?("linux")
32
+ @logger.warn("Multiple native workers require Linux for SO_REUSEPORT load balancing; use workers 0 on macOS")
33
+ end
34
+ maintain_worker_count
35
+ loop do
36
+ @signals.drain.each { |signal| handle_signal(signal) }
37
+ read_statuses
38
+ reap_children
39
+ check_timeouts
40
+ maintain_worker_count unless @shutdown_at
41
+ @reports&.write(status)
42
+ break if @shutdown_at && @workers.empty?
43
+
44
+ IO.select([@signals.io, *@workers.values.filter_map { |handle| handle.channel.io unless handle.channel.closed? }], nil, nil, 0.05)
45
+ end
46
+ @exit_status
47
+ ensure
48
+ cleanup
49
+ ForkGuard.deactivate if @guard
50
+ end
51
+
52
+ def status
53
+ { pid: Process.pid, state: @shutdown_at ? "draining" : "running", workers: @workers.values.map(&:to_h) }
54
+ end
55
+
56
+ private
57
+
58
+ def now = Process.clock_gettime(Process::CLOCK_MONOTONIC)
59
+
60
+ def maintain_worker_count
61
+ used = @workers.values.map(&:index)
62
+ (@desired - @workers.size).times do
63
+ index = (0...@desired).find { |candidate| !used.include?(candidate) }
64
+ spawn_worker(index)
65
+ used << index
66
+ end
67
+ end
68
+
69
+ def spawn_worker(index)
70
+ reader, writer = IO.pipe
71
+ @config.run_hooks(:before_fork, index)
72
+ experimental = @config.fork_mode == :grpc_fork_support
73
+ if experimental
74
+ require "gritz/native"
75
+ Transport::Native.prefork
76
+ prepared = true
77
+ end
78
+ pid = Process.fork do
79
+ exit_status = 1
80
+ begin
81
+ reader.close
82
+ # Restore all inherited master traps, including CHLD and resize signals.
83
+ @signals.close
84
+ @reports&.close
85
+ @workers.each_value(&:close)
86
+ Transport::Native.postfork_child if experimental
87
+ exit_status = Worker::Runner.new(index: index, status_io: writer, config: @config, logger: @logger).run
88
+ rescue StandardError, LoadError, SyntaxError, SystemExit => e
89
+ @logger.error("Worker #{index} failed: #{e.full_message}")
90
+ exit_status = 1
91
+ ensure
92
+ begin
93
+ writer.close unless writer.closed?
94
+ ensure
95
+ # Never unwind the inherited master ensure or run application at_exit hooks.
96
+ Process.exit!(exit_status)
97
+ end
98
+ end
99
+ end
100
+ writer.close
101
+ @workers[pid] = WorkerHandle.new(pid: pid, index: index, status_io: reader)
102
+ @logger.info("Worker #{index} spawned pid=#{pid}")
103
+ rescue StandardError
104
+ reader&.close unless reader&.closed?
105
+ writer&.close unless writer&.closed?
106
+ raise
107
+ ensure
108
+ Transport::Native.postfork_parent if prepared
109
+ end
110
+
111
+ def read_statuses
112
+ @workers.each_value do |handle|
113
+ handle.channel.read.each do |message|
114
+ handle.update(message, now: now)
115
+ if message[:state] == "failed" && !@ever_ready
116
+ @exit_status = 1
117
+ begin_shutdown(immediate: true)
118
+ end
119
+ @ever_ready = true if message[:state] == "ready"
120
+ rescue ConfigurationError => e
121
+ @logger.error("Worker #{handle.pid}: #{e.message}")
122
+ kill(handle)
123
+ end
124
+ end
125
+ end
126
+
127
+ def reap_children
128
+ # Only reap owned children: application hooks can start unrelated subprocesses.
129
+ @workers.each_key do |pid|
130
+ result = Process.waitpid2(pid, Process::WNOHANG)
131
+ next unless result
132
+
133
+ handle = @workers.delete(pid)
134
+ handle.close
135
+ child_status = result.last
136
+ @logger.info("Worker #{handle.index} exited pid=#{pid} status=#{child_status}")
137
+ startup_failed = child_status.exited? && !@ever_ready && !@shutdown_at && !handle.term_at
138
+ shutdown_failed = @shutdown_at && child_status.exited? && !child_status.success?
139
+ if handle.state != "killed" && (startup_failed || shutdown_failed)
140
+ @exit_status = 1
141
+ begin_shutdown(immediate: true) if startup_failed
142
+ end
143
+ rescue Errno::ECHILD
144
+ @workers.delete(pid)&.close
145
+ end
146
+ end
147
+
148
+ def handle_signal(signal)
149
+ case signal
150
+ when "TERM", "INT" then begin_shutdown
151
+ when "QUIT" then begin_shutdown(immediate: true)
152
+ when "TTIN"
153
+ if !@shutdown_at && @config.bind.end_with?(":0")
154
+ @logger.warn("Cannot add a reuseport worker with port 0; configure a fixed bind port")
155
+ elsif !@shutdown_at
156
+ @desired += 1
157
+ end
158
+ when "TTOU"
159
+ if !@shutdown_at && @desired > 1
160
+ @desired -= 1
161
+ retire(@workers.values.reject(&:term_at).max_by(&:index))
162
+ end
163
+ when "HUP"
164
+ @logger.reopen
165
+ @workers.each_key { |pid| send_signal("HUP", pid) }
166
+ @logger.info("Log reopened")
167
+ end
168
+ end
169
+
170
+ def begin_shutdown(immediate: false)
171
+ @shutdown_at ||= now
172
+ if immediate
173
+ @workers.each_value { |handle| kill(handle) }
174
+ else
175
+ @workers.each_value { |handle| handle.term_at ||= @shutdown_at + @config.drain_delay }
176
+ end
177
+ end
178
+
179
+ def retire(handle)
180
+ return unless handle
181
+
182
+ handle.term_at = now
183
+ handle.state = "draining"
184
+ end
185
+
186
+ def check_timeouts
187
+ timestamp = now
188
+ @workers.each_value do |handle|
189
+ next if handle.state == "killed"
190
+
191
+ if handle.term_at && timestamp >= handle.term_at && !handle.kill_at
192
+ handle.state = "draining"
193
+ send_signal("TERM", handle.pid)
194
+ handle.kill_at = timestamp + @config.shutdown_timeout
195
+ end
196
+ if handle.kill_at
197
+ kill(handle) if timestamp >= handle.kill_at
198
+ elsif !@shutdown_at
199
+ elapsed = timestamp - (handle.state == "booting" ? handle.born_at : handle.last_seen)
200
+ timeout = handle.state == "booting" ? @config.worker_boot_timeout : @config.worker_timeout
201
+ if elapsed > timeout
202
+ @logger.error("Worker #{handle.pid} #{handle.state} timeout after #{elapsed.round(2)}s")
203
+ kill(handle)
204
+ end
205
+ end
206
+ end
207
+ end
208
+
209
+ def kill(handle)
210
+ handle.state = "killed"
211
+ send_signal("KILL", handle.pid)
212
+ end
213
+
214
+ def send_signal(signal, pid)
215
+ Process.kill(signal, pid)
216
+ rescue Errno::ESRCH
217
+ nil
218
+ end
219
+
220
+ def cleanup
221
+ @workers.each_value { |handle| kill(handle) }
222
+ @workers.each_value do |handle| # rubocop:disable Style/CombinableLoops -- kill every child before waiting for any child
223
+ Process.waitpid(handle.pid)
224
+ rescue Errno::ECHILD
225
+ nil
226
+ ensure
227
+ handle.close
228
+ end
229
+ @workers.clear
230
+ @signals&.close
231
+ @reports&.close
232
+ end
233
+ end
234
+ end
235
+ end
@@ -0,0 +1,49 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Gritz
4
+ module Supervisor
5
+ # Wakes the event loop without doing process or logger work in a signal trap.
6
+ # @api private
7
+ class SignalQueue
8
+ attr_reader :io
9
+
10
+ def initialize(signals:)
11
+ @io, @writer = IO.pipe
12
+ @pending = []
13
+ @previous = {}
14
+ signals.each do |name|
15
+ @previous[name] = Signal.trap(name) do
16
+ @pending << name unless @pending.include?(name)
17
+ @writer.write_nonblock(".", exception: false)
18
+ rescue IOError, Errno::EPIPE
19
+ nil
20
+ end
21
+ end
22
+ rescue StandardError
23
+ close
24
+ raise
25
+ end
26
+
27
+ def drain
28
+ loop do
29
+ break unless @io.read_nonblock(4096, exception: false).is_a?(String)
30
+ end
31
+ # Swap instead of clearing: a signal arriving here belongs to the next drain.
32
+ pending = @pending
33
+ @pending = []
34
+ pending
35
+ end
36
+
37
+ def close
38
+ @previous&.each { |name, handler| Signal.trap(name, handler) }
39
+ @previous = {}
40
+ close_in_child
41
+ end
42
+
43
+ def close_in_child
44
+ @io&.close unless @io&.closed?
45
+ @writer&.close unless @writer&.closed?
46
+ end
47
+ end
48
+ end
49
+ end
@@ -0,0 +1,112 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+
5
+ module Gritz
6
+ module Supervisor
7
+ # Bounded, nonblocking JSON Lines over a worker's dedicated status pipe.
8
+ # @api private
9
+ class StatusChannel
10
+ # ponytail: master snapshots are capped at 64 KiB; chunk reports if clusters outgrow this bound.
11
+ MAX_LINE_BYTES = 64 * 1024
12
+ MAX_READ_BYTES = 64 * 1024
13
+
14
+ attr_reader :io
15
+
16
+ def initialize(io)
17
+ @io = io
18
+ @buffer = +""
19
+ @pending = +""
20
+ @discarding = false
21
+ @closed = io.closed?
22
+ end
23
+
24
+ def write(status)
25
+ return false if closed? || flush_pending.positive?
26
+
27
+ line = "#{JSON.generate(status)}\n"
28
+ return false if line.bytesize > MAX_LINE_BYTES
29
+
30
+ @pending = line
31
+ flush_pending.zero?
32
+ rescue JSON::GeneratorError, JSON::NestingError
33
+ false
34
+ rescue IOError, SystemCallError
35
+ close
36
+ false
37
+ end
38
+
39
+ def read
40
+ rows = []
41
+ return rows if closed?
42
+
43
+ remaining = MAX_READ_BYTES
44
+ while remaining.positive?
45
+ chunk = @io.read_nonblock([4096, remaining].min, exception: false)
46
+ break if chunk == :wait_readable
47
+
48
+ if chunk.nil?
49
+ close
50
+ break
51
+ end
52
+
53
+ remaining -= chunk.bytesize
54
+ consume(chunk, rows)
55
+ end
56
+ rows
57
+ rescue IOError, SystemCallError
58
+ close
59
+ rows
60
+ end
61
+
62
+ def closed? = @closed || @io.closed?
63
+
64
+ def close
65
+ @closed = true
66
+ @buffer.clear
67
+ @pending.clear
68
+ @io.close unless @io.closed?
69
+ nil
70
+ rescue IOError
71
+ nil
72
+ end
73
+
74
+ private
75
+
76
+ def flush_pending
77
+ return 0 if @pending.empty?
78
+
79
+ written = @io.write_nonblock(@pending, exception: false)
80
+ return @pending.bytesize if written == :wait_writable
81
+
82
+ @pending = @pending.byteslice(written..) || +""
83
+ @pending.bytesize
84
+ end
85
+
86
+ def consume(chunk, rows)
87
+ chunk.each_line do |part|
88
+ complete = part.end_with?("\n")
89
+ unless @discarding
90
+ @buffer << part
91
+ if @buffer.bytesize > MAX_LINE_BYTES
92
+ @buffer.clear
93
+ @discarding = true
94
+ end
95
+ end
96
+ next unless complete
97
+
98
+ unless @discarding
99
+ begin
100
+ row = JSON.parse(@buffer, symbolize_names: true)
101
+ rows << row if row.is_a?(Hash)
102
+ rescue JSON::ParserError
103
+ # A damaged record does not prevent subsequent status updates.
104
+ end
105
+ end
106
+ @buffer.clear
107
+ @discarding = false
108
+ end
109
+ end
110
+ end
111
+ end
112
+ end
@@ -0,0 +1,34 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Gritz
4
+ module Supervisor
5
+ # Parent-owned worker state; heartbeat deadlines use the parent's monotonic clock.
6
+ # @api private
7
+ class WorkerHandle
8
+ STATES = %w[booting ready draining failed stopped killed].freeze
9
+ attr_reader :pid, :index, :channel, :born_at, :last_seen, :stats
10
+ attr_accessor :state, :term_at, :kill_at
11
+
12
+ def initialize(pid:, index:, status_io:, now: Process.clock_gettime(Process::CLOCK_MONOTONIC))
13
+ @pid = pid
14
+ @index = index
15
+ @channel = StatusChannel.new(status_io)
16
+ @born_at = @last_seen = now
17
+ @state = "booting"
18
+ @stats = {}
19
+ end
20
+
21
+ def update(message, now:)
22
+ raise ConfigurationError, "Invalid worker state #{message[:state].inspect}" unless STATES.include?(message[:state])
23
+
24
+ @last_seen = now
25
+ @stats = message.except(:pid, :index, :state)
26
+ # A late heartbeat must not undo the parent's decision to retire this worker.
27
+ @state = message[:state] unless %w[draining killed].include?(@state)
28
+ end
29
+
30
+ def to_h = @stats.merge(pid: @pid, index: @index, state: @state)
31
+ def close = @channel.close
32
+ end
33
+ end
34
+ end
@@ -0,0 +1,150 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "rbconfig"
4
+ require "tempfile"
5
+ require "timeout"
6
+ require "io/wait"
7
+
8
+ module Gritz
9
+ module Testing
10
+ # Starts a supervisor in a fresh interpreter, avoiding gRPC state inherited from a test process.
11
+ # @api public
12
+ class Cluster
13
+ MAX_LOG_BYTES = 64 * 1024
14
+
15
+ attr_reader :pid
16
+
17
+ def self.start(**)
18
+ cluster = new(**).start
19
+ return cluster unless block_given?
20
+
21
+ begin
22
+ yield cluster
23
+ ensure
24
+ cluster.stop
25
+ end
26
+ end
27
+
28
+ def initialize(config_path:, env: {}, command: nil)
29
+ @config_path = File.expand_path(config_path)
30
+ @env = env
31
+ @command = command
32
+ @status = { state: "starting", workers: [] }
33
+ end
34
+
35
+ def start
36
+ raise ArgumentError, "cluster is already started" if @pid
37
+
38
+ @reader, writer = IO.pipe
39
+ @channel = Supervisor::StatusChannel.new(@reader)
40
+ @log = Tempfile.new(["gritz-cluster", ".log"])
41
+ command = @command || [
42
+ RbConfig.ruby, "-I", $LOAD_PATH.join(File::PATH_SEPARATOR), "-rgritz/core", "-e",
43
+ "exit Gritz::CLI.new(status_io: IO.for_fd(3)).run(ARGV)", "--", "start", "-C", @config_path
44
+ ]
45
+ @pid = Process.spawn(@env, *command, 3 => writer, out: @log, err: @log, pgroup: true)
46
+ self
47
+ rescue StandardError
48
+ unless @pid
49
+ @channel&.close
50
+ @log&.close!
51
+ end
52
+ raise
53
+ ensure
54
+ writer&.close
55
+ end
56
+
57
+ def status
58
+ @channel&.read&.each { |row| @status = row }
59
+ @status
60
+ end
61
+
62
+ def workers = status.fetch(:workers, [])
63
+
64
+ def logs
65
+ return @logs || "" unless @log && !@log.closed?
66
+
67
+ @log.flush
68
+ File.open(@log.path) do |file|
69
+ file.seek([file.size - MAX_LOG_BYTES, 0].max)
70
+ file.read
71
+ end
72
+ end
73
+
74
+ def signal(name, pid: @pid)
75
+ raise ArgumentError, "cluster is not started" unless pid
76
+
77
+ Process.kill(name, pid)
78
+ self
79
+ end
80
+
81
+ # A predicate can inspect snapshots without fixed sleeps or dependence on log wording.
82
+ def wait_until(state: "ready", workers: nil, timeout: 10)
83
+ deadline = monotonic + timeout
84
+ loop do
85
+ snapshot = status
86
+ ready = if state == "ready"
87
+ snapshot[:state] == "running" && snapshot[:workers]&.any? && snapshot[:workers].all? { |worker| worker[:state] == "ready" }
88
+ else
89
+ state.nil? || snapshot[:state] == state
90
+ end
91
+ count = workers.nil? || snapshot.fetch(:workers, []).size == workers
92
+ return self if ready && count && (!block_given? || yield(snapshot))
93
+
94
+ if exited?
95
+ raise "cluster exited #{@exit_status.inspect} before reaching #{state.inspect}\n#{logs}"
96
+ end
97
+ raise Timeout::Error, "cluster did not reach #{state.inspect}: #{snapshot.inspect}\n#{logs}" if monotonic >= deadline
98
+
99
+ pause = (deadline - monotonic).clamp(0, 0.05)
100
+ @reader.closed? ? sleep(pause) : @reader.wait_readable(pause)
101
+ end
102
+ end
103
+
104
+ def wait(timeout: 10)
105
+ raise ArgumentError, "cluster is not started" unless @pid
106
+
107
+ deadline = monotonic + timeout
108
+ until exited?
109
+ raise Timeout::Error, "cluster did not exit\n#{logs}" if monotonic >= deadline
110
+
111
+ sleep((deadline - monotonic).clamp(0, 0.01))
112
+ end
113
+ @exit_status
114
+ end
115
+
116
+ def stop(timeout: 10)
117
+ return self unless @pid
118
+
119
+ signal("TERM") unless exited?
120
+ wait(timeout:)
121
+ self
122
+ rescue Timeout::Error
123
+ begin
124
+ Process.kill("KILL", -@pid)
125
+ rescue Errno::ESRCH
126
+ # The process group may have exited between the timeout and kill.
127
+ end
128
+ wait(timeout: 5)
129
+ self
130
+ ensure
131
+ @channel&.close
132
+ @logs = logs
133
+ @log&.close!
134
+ end
135
+
136
+ private
137
+
138
+ def monotonic = Process.clock_gettime(Process::CLOCK_MONOTONIC)
139
+
140
+ def exited?
141
+ return true if @exit_status
142
+ return false unless @pid
143
+
144
+ pair = Process.waitpid2(@pid, Process::WNOHANG)
145
+ @exit_status = pair&.last
146
+ !@exit_status.nil?
147
+ end
148
+ end
149
+ end
150
+ end
@@ -0,0 +1,111 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "io/wait"
4
+
5
+ module Gritz
6
+ module Worker
7
+ # Owns transport resources and heartbeats in one serving process.
8
+ # @api private
9
+ class Runner
10
+ def initialize(index:, config:, logger:, status_io: nil)
11
+ @index = index
12
+ @config = config
13
+ @logger = logger
14
+ @status = Supervisor::StatusChannel.new(status_io) if status_io
15
+ end
16
+
17
+ def run
18
+ @exit_status = 0
19
+ begin
20
+ @signals = Supervisor::SignalQueue.new(signals: %w[TERM INT QUIT HUP])
21
+ report("booting")
22
+ require "gritz/native"
23
+ @boot_started = true
24
+ @config.preload! unless @config.preload_app?
25
+ @config.run_hooks(:on_worker_boot, @index)
26
+ router = Router.new(controllers: @config.controllers, strict: @config.strict_routes, logger: @logger)
27
+ dispatcher = Dispatcher.new(router:, middleware: @config.middleware, logger: @logger)
28
+ @adapter = Transport::Native.new(config: @config, dispatcher:, logger: @logger)
29
+ @port = @adapter.bind(@config.bind)
30
+ @adapter.start
31
+ @logger.info("Gritz #{Core::VERSION} listening on #{@config.bind.sub(/:\d+\z/, ":#{@port}")}")
32
+ report("ready")
33
+ serve
34
+ rescue StandardError, LoadError, SyntaxError, SystemExit => e
35
+ fail_worker(e)
36
+ ensure
37
+ finish
38
+ end
39
+ @exit_status
40
+ end
41
+
42
+ private
43
+
44
+ def serve
45
+ next_status = monotonic + @config.status_interval
46
+ loop do
47
+ raise ConfigurationError, "gRPC server stopped unexpectedly" unless @adapter.running?
48
+ raise ConfigurationError, "worker status pipe closed" if @status&.closed?
49
+
50
+ @signals.drain.each do |name|
51
+ case name
52
+ when "HUP" then @logger.reopen
53
+ when "QUIT"
54
+ report("draining")
55
+ @adapter.kill
56
+ @transport_stopped = true
57
+ return 0
58
+ when "TERM", "INT"
59
+ report("draining")
60
+ @adapter.stop(deadline: Time.now + @config.shutdown_timeout)
61
+ @transport_stopped = true
62
+ return 0
63
+ end
64
+ end
65
+ now = monotonic
66
+ if now >= next_status
67
+ report("ready")
68
+ next_status = now + @config.status_interval
69
+ end
70
+ @signals.io.wait_readable([next_status - monotonic, 0].max)
71
+ end
72
+ end
73
+
74
+ def report(state, include_stats: true, **extra)
75
+ return unless @status
76
+
77
+ stats = include_stats && @adapter ? @adapter.stats : {}
78
+ row = { inflight: 0, capacity: @config.threads, requests_total: 0, oldest_inflight_age: 0 }.merge(stats)
79
+ row.merge!(pid: Process.pid, index: @index, state:, ts: monotonic, port: @port,
80
+ busy_threads: stats.fetch(:busy_threads, stats.fetch(:busy, 0)))
81
+ @status.write(row.merge(extra))
82
+ end
83
+
84
+ def fail_worker(error)
85
+ @exit_status = 1
86
+ @logger.error(error.full_message)
87
+ report("failed", include_stats: false, error: error.message[0, 512])
88
+ end
89
+
90
+ def finish
91
+ cleanup { @adapter&.kill unless @transport_stopped }
92
+ cleanup { @config.run_hooks(:on_worker_shutdown, @index) if @boot_started }
93
+ report("stopped", include_stats: false, exit_status: @exit_status)
94
+ ensure
95
+ begin
96
+ @signals&.close
97
+ ensure
98
+ @status&.close
99
+ end
100
+ end
101
+
102
+ def cleanup
103
+ yield
104
+ rescue StandardError, LoadError, SyntaxError, SystemExit => e
105
+ fail_worker(e)
106
+ end
107
+
108
+ def monotonic = Process.clock_gettime(Process::CLOCK_MONOTONIC)
109
+ end
110
+ end
111
+ end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: gritz-core
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.1.0
4
+ version: 0.2.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Yudai Takada
@@ -111,6 +111,7 @@ files:
111
111
  - lib/gritz/dispatcher.rb
112
112
  - lib/gritz/dsl.rb
113
113
  - lib/gritz/errors.rb
114
+ - lib/gritz/fork_guard.rb
114
115
  - lib/gritz/method_descriptor.rb
115
116
  - lib/gritz/middleware/context.rb
116
117
  - lib/gritz/middleware/exception_mapper.rb
@@ -118,10 +119,16 @@ files:
118
119
  - lib/gritz/middleware/request_id.rb
119
120
  - lib/gritz/middleware/stack.rb
120
121
  - lib/gritz/router.rb
122
+ - lib/gritz/supervisor/master.rb
123
+ - lib/gritz/supervisor/signal_queue.rb
124
+ - lib/gritz/supervisor/status_channel.rb
125
+ - lib/gritz/supervisor/worker_handle.rb
126
+ - lib/gritz/testing/cluster.rb
121
127
  - lib/gritz/testing/in_memory_call.rb
122
128
  - lib/gritz/testing/minitest.rb
123
129
  - lib/gritz/testing/rpc_helper.rb
124
130
  - lib/gritz/testing/rspec.rb
131
+ - lib/gritz/worker/runner.rb
125
132
  homepage: https://github.com/gritzrpc/gritz-core
126
133
  licenses:
127
134
  - MIT
@@ -144,7 +151,7 @@ required_rubygems_version: !ruby/object:Gem::Requirement
144
151
  - !ruby/object:Gem::Version
145
152
  version: '0'
146
153
  requirements: []
147
- rubygems_version: 4.0.16
154
+ rubygems_version: 3.6.9
148
155
  specification_version: 4
149
156
  summary: Transport-independent controllers and middleware for Gritz
150
157
  test_files: []