wurk 1.1.0 → 1.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +25 -0
- data/app/controllers/concerns/wurk/same_origin_guard.rb +40 -0
- data/app/controllers/concerns/wurk/sse_streaming.rb +48 -0
- data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +53 -0
- data/app/controllers/wurk/api/pagination.rb +60 -11
- data/app/controllers/wurk/api/serializers.rb +5 -1
- data/app/controllers/wurk/api_controller.rb +51 -70
- data/app/controllers/wurk/application_controller.rb +26 -0
- data/app/controllers/wurk/dashboard_controller.rb +23 -1
- data/app/controllers/wurk/extensions_controller.rb +3 -12
- data/app/controllers/wurk/profiles_controller.rb +6 -1
- data/lib/wurk/batch/death_handler.rb +13 -0
- data/lib/wurk/batch/server_middleware.rb +9 -5
- data/lib/wurk/batch.rb +3 -0
- data/lib/wurk/capsule.rb +34 -21
- data/lib/wurk/cli.rb +16 -3
- data/lib/wurk/client/buffered.rb +30 -10
- data/lib/wurk/client.rb +45 -1
- data/lib/wurk/component.rb +23 -9
- data/lib/wurk/configuration.rb +83 -18
- data/lib/wurk/cron.rb +13 -1
- data/lib/wurk/dead_set.rb +16 -1
- data/lib/wurk/engine.rb +17 -1
- data/lib/wurk/fetcher/reliable.rb +46 -13
- data/lib/wurk/fetcher.rb +5 -0
- data/lib/wurk/health.rb +74 -22
- data/lib/wurk/history.rb +4 -21
- data/lib/wurk/job_retry.rb +3 -5
- data/lib/wurk/launcher.rb +26 -3
- data/lib/wurk/limiter/server_middleware.rb +16 -1
- data/lib/wurk/limiter.rb +6 -5
- data/lib/wurk/lua/loader.rb +11 -5
- data/lib/wurk/lua.rb +100 -9
- data/lib/wurk/manager.rb +59 -19
- data/lib/wurk/metrics/history.rb +24 -38
- data/lib/wurk/metrics/queue_rollup.rb +4 -21
- data/lib/wurk/metrics/rollup.rb +4 -21
- data/lib/wurk/processor.rb +8 -8
- data/lib/wurk/profile_set.rb +21 -6
- data/lib/wurk/profiler.rb +7 -5
- data/lib/wurk/rails_boot.rb +176 -0
- data/lib/wurk/railtie.rb +19 -47
- data/lib/wurk/redis_connection.rb +6 -9
- data/lib/wurk/redis_pool.rb +148 -28
- data/lib/wurk/scheduled.rb +35 -12
- data/lib/wurk/swarm/backoff.rb +70 -0
- data/lib/wurk/swarm/child_boot.rb +92 -13
- data/lib/wurk/swarm/orphan_guard.rb +105 -0
- data/lib/wurk/swarm/restart.rb +196 -0
- data/lib/wurk/swarm.rb +194 -78
- data/lib/wurk/timer_loop.rb +49 -0
- data/lib/wurk/version.rb +1 -1
- data/lib/wurk/web/extension.rb +4 -1
- data/lib/wurk/web/pool_scope.rb +46 -0
- data/lib/wurk/web/rack_app.rb +2 -1
- data/lib/wurk/web/search.rb +77 -18
- data/lib/wurk/web.rb +1 -0
- data/lib/wurk/worker/setter.rb +6 -1
- data/lib/wurk.rb +6 -2
- data/vendor/assets/dashboard/assets/ArgsValue-DYfBiXrJ.js +1 -0
- data/vendor/assets/dashboard/assets/BatchDetail-DTZ2HzcD.js +1 -0
- data/vendor/assets/dashboard/assets/Batches-BrnXA332.js +1 -0
- data/vendor/assets/dashboard/assets/Busy-C57G8Xb3.js +1 -0
- data/vendor/assets/dashboard/assets/Cron-DlyH88oo.js +1 -0
- data/vendor/assets/dashboard/assets/Dashboard-B887pxWf.js +1 -0
- data/vendor/assets/dashboard/assets/Dead-BPA7gs-X.js +1 -0
- data/vendor/assets/dashboard/assets/Extension-Bunf6XuU.js +1 -0
- data/vendor/assets/dashboard/assets/FilterBox-FCDi4ZCU.js +1 -0
- data/vendor/assets/dashboard/assets/JobDetailModal-DuMdKUMm.js +2 -0
- data/vendor/assets/dashboard/assets/Limiters-Br0aCPMK.js +1 -0
- data/vendor/assets/dashboard/assets/Metrics-DxBmuywH.js +1 -0
- data/vendor/assets/dashboard/assets/Modal-t4FI_LaY.js +1 -0
- data/vendor/assets/dashboard/assets/PageHeader-CsDvJSOA.js +1 -0
- data/vendor/assets/dashboard/assets/Profiles-Bkhoqjlq.js +1 -0
- data/vendor/assets/dashboard/assets/Queues-BvhA-vfI.js +1 -0
- data/vendor/assets/dashboard/assets/Retries-JEpB-1Yl.js +1 -0
- data/vendor/assets/dashboard/assets/Scheduled-DN_FbSwP.js +1 -0
- data/vendor/assets/dashboard/assets/Search-DJuK0YCJ.js +1 -0
- data/vendor/assets/dashboard/assets/Skeleton-DOYDkzg1.js +1 -0
- data/vendor/assets/dashboard/assets/charts-CVK0zAnC.js +1 -0
- data/vendor/assets/dashboard/assets/index-BnPX9Ptn.css +1 -0
- data/vendor/assets/dashboard/assets/index-CZTcs-pM.js +141 -0
- data/vendor/assets/dashboard/assets/useResetPageOnEmpty-CoZU4b3a.js +1 -0
- data/vendor/assets/dashboard/assets/useSort-wQcnbdsa.js +1 -0
- data/vendor/assets/dashboard/index.html +3 -3
- data/vendor/assets/dashboard/wurk-manifest.json +2 -2
- metadata +55 -26
- data/vendor/assets/dashboard/assets/ArgsValue-BUqJa-eG.js +0 -1
- data/vendor/assets/dashboard/assets/BatchDetail-C5dkqAzp.js +0 -1
- data/vendor/assets/dashboard/assets/Batches-bgkXn8tc.js +0 -1
- data/vendor/assets/dashboard/assets/Busy-QSHBFXhy.js +0 -1
- data/vendor/assets/dashboard/assets/Cron-CUHppvTA.js +0 -1
- data/vendor/assets/dashboard/assets/Dashboard-CzPudckV.js +0 -1
- data/vendor/assets/dashboard/assets/Dead-BIq4Nz_i.js +0 -1
- data/vendor/assets/dashboard/assets/Extension-CW36i9R1.js +0 -1
- data/vendor/assets/dashboard/assets/JobDetailModal-BhNdWSp7.js +0 -2
- data/vendor/assets/dashboard/assets/Limiters-CiI_DfUK.js +0 -1
- data/vendor/assets/dashboard/assets/Metrics-BzZ8ugms.js +0 -1
- data/vendor/assets/dashboard/assets/Modal-DzVfgsSF.js +0 -1
- data/vendor/assets/dashboard/assets/PageHeader-Dp3qhX3e.js +0 -1
- data/vendor/assets/dashboard/assets/Profiles-BTtIlTdR.js +0 -1
- data/vendor/assets/dashboard/assets/Queues-BuXoxQ4W.js +0 -1
- data/vendor/assets/dashboard/assets/Retries-D4HAPaOQ.js +0 -1
- data/vendor/assets/dashboard/assets/Scheduled-6NCZYVJh.js +0 -1
- data/vendor/assets/dashboard/assets/Search-JaB_-52c.js +0 -1
- data/vendor/assets/dashboard/assets/charts-6uvCyY0x.js +0 -1
- data/vendor/assets/dashboard/assets/i18n-gIeA5VLo.js +0 -1
- data/vendor/assets/dashboard/assets/index-BDG9tvBA.css +0 -1
- data/vendor/assets/dashboard/assets/index-DWfWAGBc.js +0 -141
- data/vendor/assets/dashboard/assets/useJobSetActions-DTDaAoWr.js +0 -1
- data/vendor/assets/dashboard/assets/usePageParam-BcAvRko-.js +0 -1
- data/vendor/assets/dashboard/assets/useSort-D5Am4bGq.js +0 -1
data/lib/wurk/swarm.rb
CHANGED
|
@@ -5,11 +5,20 @@ require_relative 'launcher'
|
|
|
5
5
|
require_relative 'fetcher/reliable'
|
|
6
6
|
require_relative 'keys'
|
|
7
7
|
require_relative 'swarm/child_boot'
|
|
8
|
+
require_relative 'swarm/backoff'
|
|
9
|
+
require_relative 'swarm/restart'
|
|
10
|
+
require_relative 'swarm/orphan_guard'
|
|
8
11
|
|
|
9
12
|
module Wurk
|
|
10
13
|
# Parent supervisor. Forks N children per the worker topology, monitors
|
|
11
|
-
# PIDs, relays signals, respawns crashed children
|
|
12
|
-
# restart on SIGUSR1, recycles RSS-bloated children.
|
|
14
|
+
# PIDs, relays signals, respawns crashed children with per-slot exponential
|
|
15
|
+
# backoff, handles rolling restart on SIGUSR1, recycles RSS-bloated children.
|
|
16
|
+
#
|
|
17
|
+
# The supervise loop never sleeps on behalf of a respawn or a restart: crash
|
|
18
|
+
# backoff is tracked as per-slot due-times (Swarm::Backoff) and rolling
|
|
19
|
+
# restart / recycle run as a non-blocking state machine (Swarm::Restart)
|
|
20
|
+
# advanced one phase per tick. TERM/INT is therefore honored within a tick
|
|
21
|
+
# regardless of restart or backoff state.
|
|
13
22
|
#
|
|
14
23
|
# Boot ordering (must be exact — see docs/idea/03-process-model.md):
|
|
15
24
|
# 1. Host app boots fully; eager loads done.
|
|
@@ -21,11 +30,10 @@ module Wurk
|
|
|
21
30
|
# 6. Parent calls `supervise` to enter the wait/relay loop.
|
|
22
31
|
#
|
|
23
32
|
# Signals (see docs/idea/04-signals.md):
|
|
24
|
-
# TERM/INT → `shutdown` (graceful drain)
|
|
25
|
-
# TSTP → relay TSTP (
|
|
26
|
-
# CONT → relay CONT (resume fetch)
|
|
33
|
+
# TERM/INT → `shutdown` (graceful drain; aborts any restart)
|
|
34
|
+
# TSTP → relay TSTP (quiet — stop fetching; one-way, no resume)
|
|
27
35
|
# USR1 → `rolling_restart` (zero-downtime cycle)
|
|
28
|
-
class Swarm
|
|
36
|
+
class Swarm # rubocop:disable Metrics/ClassLength
|
|
29
37
|
include Component
|
|
30
38
|
|
|
31
39
|
SUPERVISE_TICK = 0.2
|
|
@@ -34,6 +42,16 @@ module Wurk
|
|
|
34
42
|
MEMORY_CHECK_INTERVAL = 10
|
|
35
43
|
DEFAULT_SHUTDOWN_TIMEOUT = 25
|
|
36
44
|
|
|
45
|
+
# Children each hard_shutdown after their own drain deadline (bulk_requeue
|
|
46
|
+
# + a 3s ensure window + heartbeat cleanup); the parent must not SIGKILL
|
|
47
|
+
# them mid-tail, so its own wait always extends past theirs by this much.
|
|
48
|
+
SHUTDOWN_GRACE = 5
|
|
49
|
+
|
|
50
|
+
# USR2 is relayed (log reopen) — without a trap, a logrotate config that
|
|
51
|
+
# signals the master pid would hit USR2's default disposition and kill the
|
|
52
|
+
# whole swarm.
|
|
53
|
+
SWARM_SIGNALS = { 'TERM' => :term, 'INT' => :term, 'TSTP' => :tstp, 'USR1' => :usr1, 'USR2' => :usr2 }.freeze
|
|
54
|
+
|
|
37
55
|
attr_reader :topology, :children
|
|
38
56
|
|
|
39
57
|
def initialize(topology:, config: Wurk.configuration, memory_limit: config.memory_limit_kb,
|
|
@@ -45,28 +63,40 @@ module Wurk
|
|
|
45
63
|
@children = {}
|
|
46
64
|
@assignments = []
|
|
47
65
|
@stopping = false
|
|
66
|
+
@quieted = false
|
|
48
67
|
@last_memory_check = 0
|
|
49
|
-
@
|
|
68
|
+
@signal_read = nil
|
|
69
|
+
@signal_write = nil
|
|
70
|
+
@respawn_backoff = Backoff.new(base: RESPAWN_BACKOFF)
|
|
71
|
+
@restart = build_restart
|
|
50
72
|
end
|
|
51
73
|
|
|
52
74
|
# `install_signals:` is false in tests so the integration suite can
|
|
53
75
|
# drive `shutdown` / `rolling_restart` directly without poisoning the
|
|
54
76
|
# test process's signal handlers.
|
|
77
|
+
#
|
|
78
|
+
# Traps go in BEFORE fork_children: a TERM landing in the (previously
|
|
79
|
+
# post-fork) window between fork and trap installation left the parent on
|
|
80
|
+
# its default disposition — it died instantly and orphaned live, fetching
|
|
81
|
+
# children. Installed first, the trap queues the TERM and the supervise
|
|
82
|
+
# loop drains it (relaying to children) even if it arrives mid-boot.
|
|
55
83
|
def boot(install_signals: true)
|
|
56
84
|
raise 'Wurk::Swarm already booted' unless @assignments.empty?
|
|
57
85
|
raise ArgumentError, 'Topology has no slots' if @topology.empty?
|
|
58
86
|
|
|
59
87
|
@assignments = @topology.assignments.freeze
|
|
88
|
+
install_signal_handlers if install_signals
|
|
60
89
|
close_parent_sockets
|
|
61
90
|
fork_children
|
|
62
|
-
install_signal_handlers if install_signals
|
|
63
91
|
@children.keys
|
|
64
92
|
end
|
|
65
93
|
|
|
66
94
|
def supervise
|
|
67
95
|
until done?
|
|
68
96
|
drain_signals
|
|
69
|
-
|
|
97
|
+
reap_children
|
|
98
|
+
spawn_due_respawns
|
|
99
|
+
@restart.advance unless @stopping
|
|
70
100
|
check_memory_pressure
|
|
71
101
|
sleep SUPERVISE_TICK
|
|
72
102
|
end
|
|
@@ -74,32 +104,45 @@ module Wurk
|
|
|
74
104
|
|
|
75
105
|
def shutdown(timeout: @shutdown_timeout)
|
|
76
106
|
@stopping = true
|
|
107
|
+
@restart.abort
|
|
77
108
|
relay_signal('TERM')
|
|
78
|
-
wait_for_children(timeout)
|
|
109
|
+
wait_for_children(timeout + SHUTDOWN_GRACE)
|
|
79
110
|
hard_kill_stragglers
|
|
80
111
|
end
|
|
81
112
|
|
|
82
|
-
#
|
|
83
|
-
#
|
|
84
|
-
#
|
|
85
|
-
#
|
|
113
|
+
# TSTP quiet is one-way and GLOBAL (spec §21.3): it must survive respawns
|
|
114
|
+
# and memory recycles, or a quieted-but-crashed child's replacement would
|
|
115
|
+
# resume fetching mid-maintenance. The flag makes every future fork boot
|
|
116
|
+
# already-quieted (see ChildBoot start_quiet).
|
|
117
|
+
def quiet_swarm
|
|
118
|
+
@quieted = true
|
|
119
|
+
relay_signal('TSTP')
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
# SIGUSR1: queue every live child for the rolling-restart state machine,
|
|
123
|
+
# which replaces one slot at a time (spawn replacement → await its
|
|
124
|
+
# heartbeat → TERM the old child → await its drain) without blocking the
|
|
125
|
+
# supervise thread, so TERM stays responsive throughout the cycle.
|
|
86
126
|
def rolling_restart
|
|
87
|
-
@children.
|
|
88
|
-
replacement = fork_child(meta[:slot], meta[:index])
|
|
89
|
-
@children[replacement] = meta
|
|
90
|
-
unless wait_for_heartbeat(replacement)
|
|
91
|
-
logger.warn do
|
|
92
|
-
"swarm: replacement #{replacement} heartbeat not seen within #{HEARTBEAT_WAIT}s; proceeding anyway"
|
|
93
|
-
end
|
|
94
|
-
end
|
|
95
|
-
safe_kill(old_pid, 'TERM')
|
|
96
|
-
wait_pid(old_pid, @shutdown_timeout)
|
|
97
|
-
@children.delete(old_pid)
|
|
98
|
-
end
|
|
127
|
+
@restart.enqueue(@children.keys)
|
|
99
128
|
end
|
|
100
129
|
|
|
101
130
|
private
|
|
102
131
|
|
|
132
|
+
def build_restart
|
|
133
|
+
Restart.new(Restart::Config.new(
|
|
134
|
+
spawn: method(:spawn_child),
|
|
135
|
+
kill: method(:safe_kill),
|
|
136
|
+
heartbeat: method(:heartbeat_seen?),
|
|
137
|
+
describe: ->(pid) { @children[pid] },
|
|
138
|
+
now: method(:monotonic),
|
|
139
|
+
logger: logger,
|
|
140
|
+
heartbeat_wait: HEARTBEAT_WAIT,
|
|
141
|
+
drain_timeout: @shutdown_timeout + SHUTDOWN_GRACE,
|
|
142
|
+
backoff: Backoff.new(base: RESPAWN_BACKOFF)
|
|
143
|
+
))
|
|
144
|
+
end
|
|
145
|
+
|
|
103
146
|
# Step 3.
|
|
104
147
|
def close_parent_sockets
|
|
105
148
|
@config.reset_redis_pools!
|
|
@@ -117,82 +160,164 @@ module Wurk
|
|
|
117
160
|
|
|
118
161
|
# Step 4.
|
|
119
162
|
def fork_children
|
|
120
|
-
@assignments.
|
|
121
|
-
@children[fork_child(slot, idx)] = { slot: slot, index: idx }
|
|
122
|
-
end
|
|
163
|
+
@assignments.each_index { |idx| spawn_child(@assignments[idx], idx) }
|
|
123
164
|
end
|
|
124
165
|
|
|
166
|
+
# Fork one child for the slot and record its spawn time so crash backoff can
|
|
167
|
+
# tell a crash-loop (short-lived) from a healthy child that finally died.
|
|
168
|
+
# Returns the child PID; never returns in the child (ChildBoot exits).
|
|
169
|
+
def spawn_child(slot, idx)
|
|
170
|
+
pid = fork_child(slot, idx)
|
|
171
|
+
@children[pid] = { slot: slot, index: idx, spawned_at: monotonic }
|
|
172
|
+
pid
|
|
173
|
+
end
|
|
174
|
+
|
|
175
|
+
# Capture the parent PID BEFORE forking and hand it to the child: read
|
|
176
|
+
# from the parent, it is race-free even if the parent dies the instant
|
|
177
|
+
# after fork (getppid in the child could already return the reaper). The
|
|
178
|
+
# child's OrphanGuard compares live getppid against it.
|
|
125
179
|
def fork_child(slot, idx)
|
|
180
|
+
parent_pid = ::Process.pid
|
|
126
181
|
pid = ::Process.fork
|
|
127
182
|
return pid if pid
|
|
128
183
|
|
|
129
|
-
|
|
184
|
+
# Drop the parent's self-pipe first: the inherited traps still write to
|
|
185
|
+
# it, so a signal landing in the window before ChildBoot resets them
|
|
186
|
+
# would surface in the PARENT's supervise loop (an operator TERMing one
|
|
187
|
+
# child pid would drain the whole swarm). Closed, the trap write no-ops.
|
|
188
|
+
@signal_read&.close
|
|
189
|
+
@signal_write&.close
|
|
190
|
+
ChildBoot.new(@config, slot, idx, parent_pid: parent_pid, start_quiet: @quieted).run
|
|
130
191
|
exit 0 # unreachable; ChildBoot exits explicitly
|
|
131
192
|
end
|
|
132
193
|
|
|
194
|
+
# Self-pipe pattern (same as Wurk::CLI): the trap only writes the signal
|
|
195
|
+
# name to a pipe — no Thread::Queue#push, which takes a mutex a trap can
|
|
196
|
+
# deadlock against. The supervise loop polls the read end each tick.
|
|
133
197
|
def install_signal_handlers
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
::Signal.trap(sig) {
|
|
198
|
+
@signal_read, @signal_write = ::IO.pipe
|
|
199
|
+
SWARM_SIGNALS.each_key do |sig|
|
|
200
|
+
::Signal.trap(sig) { emit_signal(sig) }
|
|
201
|
+
rescue ArgumentError
|
|
202
|
+
# Platform without this signal (e.g. some JRuby builds) — skip it.
|
|
203
|
+
nil
|
|
137
204
|
end
|
|
138
205
|
end
|
|
139
206
|
|
|
207
|
+
# Non-blocking self-pipe write from trap context: a blocking `puts` could
|
|
208
|
+
# stall signal delivery if the pipe fills. `exception: false` returns
|
|
209
|
+
# :wait_writable instead of raising when full (drop the coalescible
|
|
210
|
+
# duplicate); a closed pipe during shutdown is ignored too.
|
|
211
|
+
def emit_signal(sig)
|
|
212
|
+
@signal_write.write_nonblock("#{sig}\n", exception: false)
|
|
213
|
+
rescue ::IOError, ::Errno::EPIPE, ::Errno::EBADF
|
|
214
|
+
nil
|
|
215
|
+
end
|
|
216
|
+
|
|
140
217
|
def drain_signals
|
|
141
|
-
|
|
142
|
-
sym = next_signal_symbol
|
|
143
|
-
next if sym.nil?
|
|
218
|
+
return unless @signal_read
|
|
144
219
|
|
|
145
|
-
|
|
220
|
+
while (sig = read_pending_signal)
|
|
221
|
+
case SWARM_SIGNALS[sig]
|
|
146
222
|
when :term then shutdown
|
|
147
|
-
when :tstp then
|
|
148
|
-
when :cont then relay_signal('CONT')
|
|
223
|
+
when :tstp then quiet_swarm
|
|
149
224
|
when :usr1 then rolling_restart
|
|
225
|
+
when :usr2 then relay_signal('USR2')
|
|
150
226
|
end
|
|
151
227
|
end
|
|
152
228
|
end
|
|
153
229
|
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
nil
|
|
230
|
+
# One buffered line per pending signal, non-blocking (wait_readable(0)).
|
|
231
|
+
# nil once the pipe is drained, ending the loop for this tick.
|
|
232
|
+
def read_pending_signal
|
|
233
|
+
return nil unless @signal_read.wait_readable(0)
|
|
234
|
+
|
|
235
|
+
@signal_read.gets&.strip
|
|
158
236
|
end
|
|
159
237
|
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
238
|
+
# Reap every exited child this tick (not one), so a fleet-wide death
|
|
239
|
+
# recovers in parallel rather than one child per SUPERVISE_TICK. ECHILD
|
|
240
|
+
# (momentarily no children — all crashed and awaiting backoff) is not a stop
|
|
241
|
+
# condition: the swarm only stops on an explicit TERM/INT.
|
|
242
|
+
def reap_children
|
|
243
|
+
loop do
|
|
244
|
+
pid, status = ::Process.wait2(-1, ::Process::WNOHANG)
|
|
245
|
+
break unless pid
|
|
246
|
+
|
|
247
|
+
on_child_exit(pid, status)
|
|
248
|
+
end
|
|
163
249
|
rescue Errno::ECHILD
|
|
164
|
-
|
|
250
|
+
nil
|
|
165
251
|
end
|
|
166
252
|
|
|
167
253
|
def on_child_exit(pid, status)
|
|
168
254
|
meta = @children.delete(pid)
|
|
169
255
|
return unless meta
|
|
256
|
+
return if @restart.claim_exit(pid)
|
|
170
257
|
|
|
171
258
|
if @stopping
|
|
172
259
|
logger.info { "swarm: child #{pid} exited (status=#{status.exitstatus})" }
|
|
173
260
|
else
|
|
174
|
-
|
|
175
|
-
sleep RESPAWN_BACKOFF
|
|
176
|
-
@children[fork_child(meta[:slot], meta[:index])] = meta
|
|
261
|
+
schedule_respawn(pid, status, meta)
|
|
177
262
|
end
|
|
178
263
|
end
|
|
179
264
|
|
|
265
|
+
# Arm the slot's backoff instead of sleeping the supervise thread; the next
|
|
266
|
+
# due tick respawns it. A child that lived past the reset window counts as a
|
|
267
|
+
# fresh failure (base delay), so only a genuine crash-loop escalates toward
|
|
268
|
+
# the cap.
|
|
269
|
+
def schedule_respawn(pid, status, meta)
|
|
270
|
+
idx = meta[:index]
|
|
271
|
+
delay = @respawn_backoff.fail(idx, lifetime: monotonic - meta[:spawned_at])
|
|
272
|
+
logger.warn do
|
|
273
|
+
"swarm: child #{pid} died (status=#{status.exitstatus}); respawning slot #{idx} in #{delay}s"
|
|
274
|
+
end
|
|
275
|
+
end
|
|
276
|
+
|
|
277
|
+
# Respawn any slot whose backoff window has elapsed. Runs every tick; a
|
|
278
|
+
# no-op until a scheduled respawn comes due.
|
|
279
|
+
def spawn_due_respawns
|
|
280
|
+
return if @stopping
|
|
281
|
+
|
|
282
|
+
@assignments.each_index do |idx|
|
|
283
|
+
next unless @respawn_backoff.pending?(idx) && @respawn_backoff.ready?(idx)
|
|
284
|
+
|
|
285
|
+
respawn_slot(idx)
|
|
286
|
+
end
|
|
287
|
+
end
|
|
288
|
+
|
|
289
|
+
# Consume the backoff only after the fork lands. A fork/resource failure
|
|
290
|
+
# (EAGAIN, ENOMEM) must neither drop the pending respawn nor escape the
|
|
291
|
+
# supervise loop: on failure we re-arm the backoff and leave the slot
|
|
292
|
+
# pending so the next due tick retries instead of hot-looping.
|
|
293
|
+
def respawn_slot(idx)
|
|
294
|
+
spawn_child(@assignments[idx], idx)
|
|
295
|
+
@respawn_backoff.consume(idx)
|
|
296
|
+
rescue StandardError => e
|
|
297
|
+
delay = @respawn_backoff.fail(idx)
|
|
298
|
+
logger.warn { "swarm: respawn of slot #{idx} failed (#{e.class}: #{e.message}); retrying in #{delay}s" }
|
|
299
|
+
end
|
|
300
|
+
|
|
180
301
|
def check_memory_pressure
|
|
181
302
|
return unless @memory_limit
|
|
182
303
|
|
|
183
|
-
now =
|
|
304
|
+
now = monotonic
|
|
184
305
|
return if now - @last_memory_check < MEMORY_CHECK_INTERVAL
|
|
185
306
|
|
|
186
307
|
@last_memory_check = now
|
|
187
308
|
@children.dup.each_key { |pid| recycle_if_bloated(pid) }
|
|
188
309
|
end
|
|
189
310
|
|
|
311
|
+
# Route a bloated child through the restart state machine (same path as a
|
|
312
|
+
# rolling restart) so recycle is graceful — a healthy replacement takes over
|
|
313
|
+
# before the old child is TERMed — and can't overlap a restart already in
|
|
314
|
+
# flight on the slot.
|
|
190
315
|
def recycle_if_bloated(pid)
|
|
191
316
|
rss = pid_rss_kb(pid)
|
|
192
317
|
return if rss.nil? || rss < @memory_limit
|
|
193
318
|
|
|
194
319
|
logger.warn { "swarm: child #{pid} RSS #{rss}KB >= #{@memory_limit}KB; recycling" }
|
|
195
|
-
|
|
320
|
+
@restart.enqueue([pid])
|
|
196
321
|
end
|
|
197
322
|
|
|
198
323
|
def pid_rss_kb(pid)
|
|
@@ -213,22 +338,10 @@ module Wurk
|
|
|
213
338
|
nil
|
|
214
339
|
end
|
|
215
340
|
|
|
216
|
-
def wait_pid(pid, timeout)
|
|
217
|
-
deadline = ::Process.clock_gettime(::Process::CLOCK_MONOTONIC) + timeout
|
|
218
|
-
while ::Process.clock_gettime(::Process::CLOCK_MONOTONIC) < deadline
|
|
219
|
-
return true if ::Process.wait(pid, ::Process::WNOHANG)
|
|
220
|
-
|
|
221
|
-
sleep 0.1
|
|
222
|
-
end
|
|
223
|
-
false
|
|
224
|
-
rescue Errno::ECHILD
|
|
225
|
-
true
|
|
226
|
-
end
|
|
227
|
-
|
|
228
341
|
def wait_for_children(timeout)
|
|
229
|
-
deadline =
|
|
230
|
-
while
|
|
231
|
-
|
|
342
|
+
deadline = monotonic + timeout
|
|
343
|
+
while monotonic < deadline && @children.any?
|
|
344
|
+
reap_children
|
|
232
345
|
sleep 0.1
|
|
233
346
|
end
|
|
234
347
|
end
|
|
@@ -238,21 +351,24 @@ module Wurk
|
|
|
238
351
|
@children.clear
|
|
239
352
|
end
|
|
240
353
|
|
|
241
|
-
#
|
|
242
|
-
#
|
|
243
|
-
#
|
|
244
|
-
#
|
|
245
|
-
|
|
354
|
+
# Has the child written its first heartbeat yet? One non-blocking SISMEMBER,
|
|
355
|
+
# polled by the restart state machine each tick. Identity is
|
|
356
|
+
# `<hostname>:<pid>:<nonce>`; PROCESS_NONCE is set when Component loads in
|
|
357
|
+
# the parent and inherited by every fork, so the parent computes a child's
|
|
358
|
+
# identity from its PID alone. A Redis blip returns false (not seen yet) so a
|
|
359
|
+
# transient error can't crash the supervisor — the restart deadline still
|
|
360
|
+
# forces progress.
|
|
361
|
+
def heartbeat_seen?(pid)
|
|
246
362
|
identity = "#{hostname}:#{pid}:#{Component::PROCESS_NONCE}"
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
return true if @config.redis { |c| c.call('SISMEMBER', Keys::PROCESSES, identity) } == 1
|
|
250
|
-
|
|
251
|
-
sleep 0.5
|
|
252
|
-
end
|
|
363
|
+
@config.redis { |c| c.call('SISMEMBER', Keys::PROCESSES, identity) } == 1
|
|
364
|
+
rescue StandardError
|
|
253
365
|
false
|
|
254
366
|
end
|
|
255
367
|
|
|
368
|
+
def monotonic
|
|
369
|
+
::Process.clock_gettime(::Process::CLOCK_MONOTONIC)
|
|
370
|
+
end
|
|
371
|
+
|
|
256
372
|
def done?
|
|
257
373
|
@stopping && @children.empty?
|
|
258
374
|
end
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Wurk
|
|
4
|
+
# Mutex/CV "tick every N seconds until told to stop" primitive shared by
|
|
5
|
+
# every leader-gated periodic component (History, Metrics::Rollup,
|
|
6
|
+
# Metrics::QueueRollup). Each of those used to hand-roll the same
|
|
7
|
+
# `@mutex`/`@sleeper`/`@done` dance; this is that dance, extracted once.
|
|
8
|
+
#
|
|
9
|
+
# The host owns thread spawning (it needs its own `safe_thread` from
|
|
10
|
+
# Component for logger/handle_exception context) and the leader-gate check
|
|
11
|
+
# inside its `tick` — TimerLoop only owns the interval wait and the
|
|
12
|
+
# start/terminate signaling around it:
|
|
13
|
+
#
|
|
14
|
+
# @timer = Wurk::TimerLoop.new(interval)
|
|
15
|
+
# @thread ||= safe_thread('my-loop') { @timer.run { tick } }
|
|
16
|
+
# def terminate = @timer.terminate
|
|
17
|
+
class TimerLoop
|
|
18
|
+
def initialize(interval)
|
|
19
|
+
@interval = interval
|
|
20
|
+
@done = false
|
|
21
|
+
@mutex = ::Mutex.new
|
|
22
|
+
@sleeper = ::ConditionVariable.new
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
# Waits one interval, then yields repeatedly (waiting between calls)
|
|
26
|
+
# until #terminate is called. Matches the existing components' "don't
|
|
27
|
+
# tick immediately on boot" behavior.
|
|
28
|
+
def run
|
|
29
|
+
wait
|
|
30
|
+
until @done
|
|
31
|
+
yield
|
|
32
|
+
wait
|
|
33
|
+
end
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
def terminate
|
|
37
|
+
@mutex.synchronize do
|
|
38
|
+
@done = true
|
|
39
|
+
@sleeper.signal
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
def wait
|
|
44
|
+
@mutex.synchronize do
|
|
45
|
+
@sleeper.wait(@mutex, @interval) unless @done
|
|
46
|
+
end
|
|
47
|
+
end
|
|
48
|
+
end
|
|
49
|
+
end
|
data/lib/wurk/version.rb
CHANGED
data/lib/wurk/web/extension.rb
CHANGED
|
@@ -271,7 +271,10 @@ module Wurk
|
|
|
271
271
|
end
|
|
272
272
|
rescue ::StandardError => e
|
|
273
273
|
::Wurk.configuration.handle_exception(e, context: 'web-extension-render')
|
|
274
|
-
|
|
274
|
+
# Message stays server-side (handle_exception logs it): exception
|
|
275
|
+
# text can carry file paths or Redis connection details, and every
|
|
276
|
+
# dashboard viewer sees this body.
|
|
277
|
+
[500, html_headers, 'Extension render error (see server logs)']
|
|
275
278
|
end
|
|
276
279
|
|
|
277
280
|
def action_for(ext, route_params, env, ctx)
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Wurk
|
|
4
|
+
class Web
|
|
5
|
+
# Routes `Wurk.redis` onto the dedicated web pool (Wurk::Configuration
|
|
6
|
+
# #web_redis_pool) for the duration of a block.
|
|
7
|
+
#
|
|
8
|
+
# The dashboard, JSON API, and SSE stream run in the host's web process and
|
|
9
|
+
# reach Redis through the same inspector objects (Stats, Queue, Search, …) a
|
|
10
|
+
# worker uses — all of which call `Wurk.redis`. Left on the default pool, a
|
|
11
|
+
# burst of dashboard traffic or a long-held SSE stream could drain the
|
|
12
|
+
# connections a co-located (embedded) worker needs to make progress, and
|
|
13
|
+
# vice versa: the #101 pool-exhaustion incident.
|
|
14
|
+
#
|
|
15
|
+
# `Wurk.redis_pool` resolves its pool by calling `#redis_pool` on
|
|
16
|
+
# `Thread.current[:wurk_capsule]` (falling back to the default capsule).
|
|
17
|
+
# Pointing that thread-local at a handle whose `#redis_pool` is the web pool
|
|
18
|
+
# diverts every nested `Wurk.redis` call with no per-call-site change. The
|
|
19
|
+
# prior value is restored on exit so a reused web-server thread — or a
|
|
20
|
+
# nested scope — is left untouched.
|
|
21
|
+
module PoolScope
|
|
22
|
+
class << self
|
|
23
|
+
def scope
|
|
24
|
+
prev = Thread.current[:wurk_capsule]
|
|
25
|
+
Thread.current[:wurk_capsule] = handle
|
|
26
|
+
yield
|
|
27
|
+
ensure
|
|
28
|
+
Thread.current[:wurk_capsule] = prev
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
def handle
|
|
32
|
+
@handle ||= Handle.new
|
|
33
|
+
end
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
# Minimal capsule-shaped duck type: Wurk.redis_pool only ever calls
|
|
37
|
+
# `#redis_pool` on the thread-local. Resolved lazily so a post-fork
|
|
38
|
+
# `reset_redis_pools!` rebuild is picked up transparently.
|
|
39
|
+
class Handle
|
|
40
|
+
def redis_pool
|
|
41
|
+
Wurk.configuration.web_redis_pool
|
|
42
|
+
end
|
|
43
|
+
end
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
end
|
data/lib/wurk/web/rack_app.rb
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require_relative 'extension'
|
|
4
|
+
require_relative 'pool_scope'
|
|
4
5
|
|
|
5
6
|
module Wurk
|
|
6
7
|
class Web
|
|
@@ -29,7 +30,7 @@ module Wurk
|
|
|
29
30
|
# auth here would break `run Sidekiq::Web` / rack-test parity with
|
|
30
31
|
# upstream Sidekiq, which also doesn't auth-gate `Sidekiq::Web.call`.
|
|
31
32
|
def call(env)
|
|
32
|
-
config.rack_app(method(:dispatch)).call(env)
|
|
33
|
+
PoolScope.scope { config.rack_app(method(:dispatch)).call(env) }
|
|
33
34
|
end
|
|
34
35
|
|
|
35
36
|
private
|