gritz-core 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,186 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "rbconfig"
4
+ require "tempfile"
5
+ require "timeout"
6
+ require "io/wait"
7
+
8
+ module Gritz
9
+ module Testing
10
+ # Starts a supervisor in a fresh interpreter, avoiding gRPC state inherited from a test process.
11
+ # @api public
12
+ class Cluster
13
+ MAX_LOG_BYTES = 64 * 1024
14
+
15
+ attr_reader :pid
16
+
17
+ def self.start(**)
18
+ cluster = new(**).start
19
+ return cluster unless block_given?
20
+
21
+ begin
22
+ yield cluster
23
+ ensure
24
+ cluster.stop
25
+ end
26
+ end
27
+
28
+ def initialize(config_path:, env: {}, command: nil)
29
+ @config_path = File.expand_path(config_path)
30
+ @env = env
31
+ @command = command
32
+ @status = { state: "starting", workers: [] }
33
+ end
34
+
35
+ def start
36
+ raise ArgumentError, "cluster is already started" if @pid
37
+
38
+ Supervisor::Launcher.enable_subreaper!
39
+ @reader, writer = IO.pipe
40
+ @channel = Supervisor::StatusChannel.new(@reader)
41
+ @log = Tempfile.new(["gritz-cluster", ".log"])
42
+ command = @command || [
43
+ RbConfig.ruby, "-I", $LOAD_PATH.join(File::PATH_SEPARATOR), "-rgritz/core", "-e",
44
+ "exit Gritz::CLI.new(status_io: IO.for_fd(3), launch: true).run(ARGV)", "--", "start", "-C", @config_path
45
+ ]
46
+ @pid = Process.spawn(@env, *command, 3 => writer, out: @log, err: @log, pgroup: true)
47
+ self
48
+ rescue StandardError
49
+ unless @pid
50
+ @channel&.close
51
+ @log&.close!
52
+ end
53
+ raise
54
+ ensure
55
+ writer&.close
56
+ end
57
+
58
+ def status
59
+ @channel&.read&.each { |row| @status = row }
60
+ @status
61
+ end
62
+
63
+ def workers = status.fetch(:workers, [])
64
+
65
+ def master_pid = status[:pid] || @pid
66
+
67
+ def logs
68
+ return @logs || "" unless @log && !@log.closed?
69
+
70
+ @log.flush
71
+ File.open(@log.path) do |file|
72
+ file.seek([file.size - MAX_LOG_BYTES, 0].max)
73
+ file.read
74
+ end
75
+ end
76
+
77
+ def signal(name, pid: @pid)
78
+ raise ArgumentError, "cluster is not started" unless pid
79
+
80
+ Process.kill(name, pid)
81
+ self
82
+ end
83
+
84
+ # A predicate can inspect snapshots without fixed sleeps or dependence on log wording.
85
+ def wait_until(state: "ready", workers: nil, timeout: 10)
86
+ deadline = monotonic + timeout
87
+ loop do
88
+ snapshot = status
89
+ active_workers = snapshot.fetch(:workers, []).reject { |worker| worker[:retiring] }
90
+ ready = if state == "ready"
91
+ snapshot[:state] == "running" && active_workers.any? && active_workers.all? { |worker| worker[:state] == "ready" }
92
+ else
93
+ state.nil? || snapshot[:state] == state
94
+ end
95
+ count = workers.nil? || active_workers.size == workers
96
+ return self if ready && count && (!block_given? || yield(snapshot))
97
+
98
+ if exited?
99
+ raise "cluster exited #{@exit_status.inspect} before reaching #{state.inspect}\n#{logs}"
100
+ end
101
+ raise Timeout::Error, "cluster did not reach #{state.inspect}: #{snapshot.inspect}\n#{logs}" if monotonic >= deadline
102
+
103
+ pause = (deadline - monotonic).clamp(0, 0.05)
104
+ @reader.closed? ? sleep(pause) : @reader.wait_readable(pause)
105
+ end
106
+ end
107
+
108
+ def wait(timeout: 10)
109
+ raise ArgumentError, "cluster is not started" unless @pid
110
+
111
+ deadline = monotonic + timeout
112
+ until exited?
113
+ raise Timeout::Error, "cluster did not exit\n#{logs}" if monotonic >= deadline
114
+
115
+ sleep((deadline - monotonic).clamp(0, 0.01))
116
+ end
117
+ @exit_status
118
+ end
119
+
120
+ def stop(timeout: 10)
121
+ return self unless @pid
122
+
123
+ signal("TERM") unless exited?
124
+ wait(timeout:)
125
+ self
126
+ rescue Timeout::Error
127
+ signal("QUIT") if status[:owner_pid] && !exited?
128
+ begin
129
+ wait(timeout: 0.5)
130
+ rescue Timeout::Error
131
+ owned_groups.each { |pid| kill_group(pid) }
132
+ wait(timeout: 5)
133
+ end
134
+ self
135
+ ensure
136
+ cleanup_groups if @pid && @exit_status
137
+ @channel&.close
138
+ @logs = logs
139
+ @log&.close!
140
+ end
141
+
142
+ private
143
+
144
+ def monotonic = Process.clock_gettime(Process::CLOCK_MONOTONIC)
145
+
146
+ def owned_groups
147
+ [@pid, *status.fetch(:masters, []).map { |master| master[:pid] }].compact.uniq
148
+ end
149
+
150
+ def kill_group(pid)
151
+ Process.kill("KILL", -pid)
152
+ rescue Errno::ESRCH
153
+ nil
154
+ rescue Errno::EPERM
155
+ # Darwin may report EPERM for an already-disappeared process group.
156
+ begin
157
+ Process.kill(0, pid)
158
+ rescue Errno::ESRCH
159
+ return
160
+ end
161
+ raise
162
+ end
163
+
164
+ def cleanup_groups
165
+ groups = owned_groups.reject { |pid| pid == @pid }
166
+ groups.each { |pid| kill_group(pid) }
167
+ # Terminate every group before waiting for any adopted children.
168
+ groups.each do |pid| # rubocop:disable Style/CombinableLoops
169
+ loop { Process.waitpid(-pid) }
170
+ rescue Errno::ECHILD
171
+ nil
172
+ end
173
+ @status = @status.merge(masters: [])
174
+ end
175
+
176
+ def exited?
177
+ return true if @exit_status
178
+ return false unless @pid
179
+
180
+ pair = Process.waitpid2(@pid, Process::WNOHANG)
181
+ @exit_status = pair&.last
182
+ !@exit_status.nil?
183
+ end
184
+ end
185
+ end
186
+ end
@@ -0,0 +1,203 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "io/wait"
4
+
5
+ module Gritz
6
+ module Worker
7
+ # Owns transport resources and heartbeats in one serving process.
8
+ # @api private
9
+ class Runner
10
+ def initialize(index:, config:, logger:, status_io: nil, owner_channel: nil)
11
+ @index = index
12
+ @config = config
13
+ @logger = logger
14
+ @owner_channel = owner_channel
15
+ @status = owner_channel || (Supervisor::StatusChannel.new(status_io) if status_io)
16
+ @recorder = Metrics::Recorder.new
17
+ @metrics = Metrics::Aggregator.new
18
+ @metric_seq = 0
19
+ @born_at = monotonic
20
+ end
21
+
22
+ def run
23
+ @exit_status = 0
24
+ begin
25
+ @signals = Supervisor::SignalQueue.new(signals: %w[TERM INT QUIT HUP USR1 USR2])
26
+ report("booting")
27
+ require "gritz/native"
28
+ @boot_started = true
29
+ @config.preload! unless @config.preload_app?
30
+ @config.run_hooks(:on_worker_boot, @index)
31
+ router = Router.new(controllers: @config.controllers, strict: @config.strict_routes, logger: @logger)
32
+ dispatcher = Dispatcher.new(router:, middleware: @config.middleware, logger: @logger, metrics: @recorder,
33
+ worker: @index, log_format: @config.log_format, log_redact: @config.log_redact)
34
+ @adapter = Transport::Native.new(config: @config, dispatcher:, logger: @logger)
35
+ @port = @adapter.bind(@config.bind)
36
+ @adapter.start
37
+ unless @status
38
+ @admin = Supervisor::AdminServer.new(bind: @config.admin_bind, status: -> { snapshot }, ready: -> { ready? },
39
+ metrics: -> { @metrics.render(workers: [@last_status].compact) }, logger: @logger)
40
+ end
41
+ @logger.info("Gritz #{Core::VERSION} listening on #{@config.bind.sub(/:\d+\z/, ":#{@port}")}")
42
+ report("ready")
43
+ serve
44
+ rescue StandardError, LoadError, SyntaxError, SystemExit => e
45
+ fail_worker(e)
46
+ ensure
47
+ finish
48
+ end
49
+ @exit_status
50
+ end
51
+
52
+ private
53
+
54
+ def serve
55
+ next_status = monotonic + @config.status_interval
56
+ loop do
57
+ raise ConfigurationError, "gRPC server stopped unexpectedly" unless @adapter.running?
58
+ raise ConfigurationError, "worker status pipe closed" if @status&.closed?
59
+
60
+ @signals.drain.each do |name|
61
+ case name
62
+ when "HUP" then @logger.reopen
63
+ when "USR1"
64
+ if @config.workers.zero?
65
+ @logger.warn("USR1 requires workers > 0; use USR2 to reload a single-process server")
66
+ else
67
+ drain
68
+ end
69
+ when "USR2"
70
+ if @owner_channel
71
+ @pending_reexec = true
72
+ else
73
+ @logger.warn("USR2 requires the gritz launcher")
74
+ end
75
+ when "QUIT"
76
+ drain
77
+ @adapter.kill
78
+ @transport_stopped = true
79
+ return 0
80
+ when "TERM", "INT"
81
+ drain
82
+ # Supervised children have already waited in the master's drain phase.
83
+ @drain_until ||= monotonic + (@config.workers.zero? ? @config.drain_delay : 0)
84
+ end
85
+ end
86
+ now = monotonic
87
+ if @drain_until && now >= @drain_until
88
+ @stop_deadline = now + @config.shutdown_timeout
89
+ @adapter.stop(deadline: Time.now + @config.shutdown_timeout)
90
+ @transport_stopped = true
91
+ return 0
92
+ end
93
+ @status&.flush
94
+ if @pending_reexec && @owner_channel.write(type: "reexec", pid: Process.pid)
95
+ @pending_reexec = false
96
+ end
97
+ publish_metrics
98
+ @admin&.poll
99
+ if now >= next_status
100
+ report(@draining ? "draining" : "ready")
101
+ next_status = now + @config.status_interval
102
+ end
103
+ next_event = [next_status, @drain_until].compact.min
104
+ IO.select([@signals.io, *@admin&.ios.to_a], nil, nil, (next_event - monotonic).clamp(0, 0.05))
105
+ end
106
+ end
107
+
108
+ def report(state, include_stats: true, **extra)
109
+ stats = include_stats && @adapter ? @adapter.stats : {}
110
+ @recorder.observe_rejected(stats.fetch(:rejected_total, 0)) if include_stats && @adapter
111
+ evaluate_health if state == "ready"
112
+ row = { inflight: 0, capacity: @config.threads, requests_total: 0, oldest_inflight_age: 0 }.merge(stats)
113
+ row.merge!(pid: Process.pid, index: @index, state:, ts: monotonic, port: @port,
114
+ busy_threads: stats.fetch(:busy_threads, stats.fetch(:busy, 0)), healthy: state == "ready" && @healthy,
115
+ checks: @checks || {}, worker_started_at: @born_at)
116
+ @last_status = row.merge(extra)
117
+ @status&.write(@owner_channel ? snapshot : @last_status)
118
+ end
119
+
120
+ def snapshot
121
+ { type: "status", pid: Process.pid, state: @draining ? "draining" : "running", desired: 1, workers: [@last_status].compact }
122
+ end
123
+
124
+ def ready? = !@draining && @last_status && @last_status[:state] == "ready" && @last_status[:healthy]
125
+
126
+ def evaluate_health
127
+ @checks = @config.health_checks.to_h do |name, callback|
128
+ [name, callback.call ? true : false]
129
+ rescue StandardError => e
130
+ @logger.warn("Health check #{name}: #{e.message}")
131
+ [name, false]
132
+ end
133
+ @healthy = @adapter.update_health(ready: true, checks: @checks)
134
+ end
135
+
136
+ def drain
137
+ return if @draining
138
+
139
+ @draining = true
140
+ @adapter.drain!
141
+ report("draining")
142
+ end
143
+
144
+ def publish_metrics
145
+ @pending_delta ||= @recorder.take_delta
146
+ return true unless @pending_delta
147
+
148
+ packet = { type: "metrics", pid: Process.pid, worker_pid: Process.pid, worker_started_at: @born_at,
149
+ seq: @metric_seq + 1, delta: @pending_delta }
150
+ accepted = @status ? @status.write(packet) : @metrics.apply(self, packet)
151
+ if accepted
152
+ @metric_seq += 1
153
+ @pending_delta = nil
154
+ end
155
+ accepted
156
+ end
157
+
158
+ def fail_worker(error)
159
+ @exit_status = 1
160
+ @logger.error(error.full_message)
161
+ report("failed", include_stats: false, error: error.message[0, 512])
162
+ end
163
+
164
+ def finish
165
+ cleanup { @adapter&.kill unless @transport_stopped }
166
+ cleanup { @config.run_hooks(:on_worker_shutdown, @index) if @boot_started }
167
+ cleanup { @recorder.observe_rejected(@adapter.stats.fetch(:rejected_total, 0)) if @adapter }
168
+ deadline = @stop_deadline || (monotonic + @config.shutdown_timeout)
169
+ loop do
170
+ @status&.flush
171
+ accepted = publish_metrics
172
+ @pending_delta ||= @recorder.take_delta if accepted
173
+ break if accepted && !@pending_delta && (@status.nil? || @status.flush)
174
+ break if monotonic >= deadline || @status&.closed?
175
+
176
+ @status&.io&.wait_writable((deadline - monotonic).clamp(0, 0.01))
177
+ end
178
+ until report("stopped", include_stats: false, exit_status: @exit_status) != false
179
+ break if monotonic >= deadline || @status&.closed?
180
+
181
+ @status.flush
182
+ @status.io.wait_writable((deadline - monotonic).clamp(0, 0.01)) unless @status.closed?
183
+ end
184
+ @status.io.wait_writable((deadline - monotonic).clamp(0, 0.01)) until !@status || @status.flush || @status.closed? || monotonic >= deadline
185
+ ensure
186
+ begin
187
+ @signals&.close
188
+ @admin&.close
189
+ ensure
190
+ @status&.close
191
+ end
192
+ end
193
+
194
+ def cleanup
195
+ yield
196
+ rescue StandardError, LoadError, SyntaxError, SystemExit => e
197
+ fail_worker(e)
198
+ end
199
+
200
+ def monotonic = Process.clock_gettime(Process::CLOCK_MONOTONIC)
201
+ end
202
+ end
203
+ end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: gritz-core
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.1.0
4
+ version: 0.3.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Yudai Takada
@@ -9,6 +9,26 @@ bindir: bin
9
9
  cert_chain: []
10
10
  date: 1980-01-02 00:00:00.000000000 Z
11
11
  dependencies:
12
+ - !ruby/object:Gem::Dependency
13
+ name: fiddle
14
+ requirement: !ruby/object:Gem::Requirement
15
+ requirements:
16
+ - - ">="
17
+ - !ruby/object:Gem::Version
18
+ version: '1.1'
19
+ - - "<"
20
+ - !ruby/object:Gem::Version
21
+ version: '3'
22
+ type: :runtime
23
+ prerelease: false
24
+ version_requirements: !ruby/object:Gem::Requirement
25
+ requirements:
26
+ - - ">="
27
+ - !ruby/object:Gem::Version
28
+ version: '1.1'
29
+ - - "<"
30
+ - !ruby/object:Gem::Version
31
+ version: '3'
12
32
  - !ruby/object:Gem::Dependency
13
33
  name: google-protobuf
14
34
  requirement: !ruby/object:Gem::Requirement
@@ -111,17 +131,29 @@ files:
111
131
  - lib/gritz/dispatcher.rb
112
132
  - lib/gritz/dsl.rb
113
133
  - lib/gritz/errors.rb
134
+ - lib/gritz/fork_guard.rb
114
135
  - lib/gritz/method_descriptor.rb
136
+ - lib/gritz/metrics/aggregator.rb
137
+ - lib/gritz/metrics/recorder.rb
115
138
  - lib/gritz/middleware/context.rb
116
139
  - lib/gritz/middleware/exception_mapper.rb
117
140
  - lib/gritz/middleware/logging.rb
141
+ - lib/gritz/middleware/metrics.rb
118
142
  - lib/gritz/middleware/request_id.rb
119
143
  - lib/gritz/middleware/stack.rb
120
144
  - lib/gritz/router.rb
145
+ - lib/gritz/supervisor/admin_server.rb
146
+ - lib/gritz/supervisor/launcher.rb
147
+ - lib/gritz/supervisor/master.rb
148
+ - lib/gritz/supervisor/signal_queue.rb
149
+ - lib/gritz/supervisor/status_channel.rb
150
+ - lib/gritz/supervisor/worker_handle.rb
151
+ - lib/gritz/testing/cluster.rb
121
152
  - lib/gritz/testing/in_memory_call.rb
122
153
  - lib/gritz/testing/minitest.rb
123
154
  - lib/gritz/testing/rpc_helper.rb
124
155
  - lib/gritz/testing/rspec.rb
156
+ - lib/gritz/worker/runner.rb
125
157
  homepage: https://github.com/gritzrpc/gritz-core
126
158
  licenses:
127
159
  - MIT
@@ -144,7 +176,7 @@ required_rubygems_version: !ruby/object:Gem::Requirement
144
176
  - !ruby/object:Gem::Version
145
177
  version: '0'
146
178
  requirements: []
147
- rubygems_version: 4.0.16
179
+ rubygems_version: 3.6.9
148
180
  specification_version: 4
149
181
  summary: Transport-independent controllers and middleware for Gritz
150
182
  test_files: []