wurk 1.1.0 → 1.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +25 -0
  3. data/app/controllers/concerns/wurk/same_origin_guard.rb +40 -0
  4. data/app/controllers/concerns/wurk/sse_streaming.rb +48 -0
  5. data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +53 -0
  6. data/app/controllers/wurk/api/pagination.rb +60 -11
  7. data/app/controllers/wurk/api/serializers.rb +5 -1
  8. data/app/controllers/wurk/api_controller.rb +51 -70
  9. data/app/controllers/wurk/application_controller.rb +26 -0
  10. data/app/controllers/wurk/dashboard_controller.rb +23 -1
  11. data/app/controllers/wurk/extensions_controller.rb +3 -12
  12. data/app/controllers/wurk/profiles_controller.rb +6 -1
  13. data/lib/wurk/batch/death_handler.rb +13 -0
  14. data/lib/wurk/batch/server_middleware.rb +9 -5
  15. data/lib/wurk/batch.rb +3 -0
  16. data/lib/wurk/capsule.rb +34 -21
  17. data/lib/wurk/cli.rb +16 -3
  18. data/lib/wurk/client/buffered.rb +30 -10
  19. data/lib/wurk/client.rb +45 -1
  20. data/lib/wurk/component.rb +23 -9
  21. data/lib/wurk/configuration.rb +83 -18
  22. data/lib/wurk/cron.rb +13 -1
  23. data/lib/wurk/dead_set.rb +16 -1
  24. data/lib/wurk/engine.rb +17 -1
  25. data/lib/wurk/fetcher/reliable.rb +46 -13
  26. data/lib/wurk/fetcher.rb +5 -0
  27. data/lib/wurk/health.rb +74 -22
  28. data/lib/wurk/history.rb +4 -21
  29. data/lib/wurk/job_retry.rb +3 -5
  30. data/lib/wurk/launcher.rb +26 -3
  31. data/lib/wurk/limiter/server_middleware.rb +16 -1
  32. data/lib/wurk/limiter.rb +6 -5
  33. data/lib/wurk/lua/loader.rb +11 -5
  34. data/lib/wurk/lua.rb +100 -9
  35. data/lib/wurk/manager.rb +59 -19
  36. data/lib/wurk/metrics/history.rb +24 -38
  37. data/lib/wurk/metrics/queue_rollup.rb +4 -21
  38. data/lib/wurk/metrics/rollup.rb +4 -21
  39. data/lib/wurk/processor.rb +8 -8
  40. data/lib/wurk/profile_set.rb +21 -6
  41. data/lib/wurk/profiler.rb +7 -5
  42. data/lib/wurk/rails_boot.rb +176 -0
  43. data/lib/wurk/railtie.rb +19 -47
  44. data/lib/wurk/redis_connection.rb +6 -9
  45. data/lib/wurk/redis_pool.rb +148 -28
  46. data/lib/wurk/scheduled.rb +35 -12
  47. data/lib/wurk/swarm/backoff.rb +70 -0
  48. data/lib/wurk/swarm/child_boot.rb +92 -13
  49. data/lib/wurk/swarm/orphan_guard.rb +105 -0
  50. data/lib/wurk/swarm/restart.rb +196 -0
  51. data/lib/wurk/swarm.rb +194 -78
  52. data/lib/wurk/timer_loop.rb +49 -0
  53. data/lib/wurk/version.rb +1 -1
  54. data/lib/wurk/web/extension.rb +4 -1
  55. data/lib/wurk/web/pool_scope.rb +46 -0
  56. data/lib/wurk/web/rack_app.rb +2 -1
  57. data/lib/wurk/web/search.rb +77 -18
  58. data/lib/wurk/web.rb +1 -0
  59. data/lib/wurk/worker/setter.rb +6 -1
  60. data/lib/wurk.rb +6 -2
  61. data/vendor/assets/dashboard/assets/ArgsValue-DYfBiXrJ.js +1 -0
  62. data/vendor/assets/dashboard/assets/BatchDetail-DTZ2HzcD.js +1 -0
  63. data/vendor/assets/dashboard/assets/Batches-BrnXA332.js +1 -0
  64. data/vendor/assets/dashboard/assets/Busy-C57G8Xb3.js +1 -0
  65. data/vendor/assets/dashboard/assets/Cron-DlyH88oo.js +1 -0
  66. data/vendor/assets/dashboard/assets/Dashboard-B887pxWf.js +1 -0
  67. data/vendor/assets/dashboard/assets/Dead-BPA7gs-X.js +1 -0
  68. data/vendor/assets/dashboard/assets/Extension-Bunf6XuU.js +1 -0
  69. data/vendor/assets/dashboard/assets/FilterBox-FCDi4ZCU.js +1 -0
  70. data/vendor/assets/dashboard/assets/JobDetailModal-DuMdKUMm.js +2 -0
  71. data/vendor/assets/dashboard/assets/Limiters-Br0aCPMK.js +1 -0
  72. data/vendor/assets/dashboard/assets/Metrics-DxBmuywH.js +1 -0
  73. data/vendor/assets/dashboard/assets/Modal-t4FI_LaY.js +1 -0
  74. data/vendor/assets/dashboard/assets/PageHeader-CsDvJSOA.js +1 -0
  75. data/vendor/assets/dashboard/assets/Profiles-Bkhoqjlq.js +1 -0
  76. data/vendor/assets/dashboard/assets/Queues-BvhA-vfI.js +1 -0
  77. data/vendor/assets/dashboard/assets/Retries-JEpB-1Yl.js +1 -0
  78. data/vendor/assets/dashboard/assets/Scheduled-DN_FbSwP.js +1 -0
  79. data/vendor/assets/dashboard/assets/Search-DJuK0YCJ.js +1 -0
  80. data/vendor/assets/dashboard/assets/Skeleton-DOYDkzg1.js +1 -0
  81. data/vendor/assets/dashboard/assets/charts-CVK0zAnC.js +1 -0
  82. data/vendor/assets/dashboard/assets/index-BnPX9Ptn.css +1 -0
  83. data/vendor/assets/dashboard/assets/index-CZTcs-pM.js +141 -0
  84. data/vendor/assets/dashboard/assets/useResetPageOnEmpty-CoZU4b3a.js +1 -0
  85. data/vendor/assets/dashboard/assets/useSort-wQcnbdsa.js +1 -0
  86. data/vendor/assets/dashboard/index.html +3 -3
  87. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  88. metadata +55 -26
  89. data/vendor/assets/dashboard/assets/ArgsValue-BUqJa-eG.js +0 -1
  90. data/vendor/assets/dashboard/assets/BatchDetail-C5dkqAzp.js +0 -1
  91. data/vendor/assets/dashboard/assets/Batches-bgkXn8tc.js +0 -1
  92. data/vendor/assets/dashboard/assets/Busy-QSHBFXhy.js +0 -1
  93. data/vendor/assets/dashboard/assets/Cron-CUHppvTA.js +0 -1
  94. data/vendor/assets/dashboard/assets/Dashboard-CzPudckV.js +0 -1
  95. data/vendor/assets/dashboard/assets/Dead-BIq4Nz_i.js +0 -1
  96. data/vendor/assets/dashboard/assets/Extension-CW36i9R1.js +0 -1
  97. data/vendor/assets/dashboard/assets/JobDetailModal-BhNdWSp7.js +0 -2
  98. data/vendor/assets/dashboard/assets/Limiters-CiI_DfUK.js +0 -1
  99. data/vendor/assets/dashboard/assets/Metrics-BzZ8ugms.js +0 -1
  100. data/vendor/assets/dashboard/assets/Modal-DzVfgsSF.js +0 -1
  101. data/vendor/assets/dashboard/assets/PageHeader-Dp3qhX3e.js +0 -1
  102. data/vendor/assets/dashboard/assets/Profiles-BTtIlTdR.js +0 -1
  103. data/vendor/assets/dashboard/assets/Queues-BuXoxQ4W.js +0 -1
  104. data/vendor/assets/dashboard/assets/Retries-D4HAPaOQ.js +0 -1
  105. data/vendor/assets/dashboard/assets/Scheduled-6NCZYVJh.js +0 -1
  106. data/vendor/assets/dashboard/assets/Search-JaB_-52c.js +0 -1
  107. data/vendor/assets/dashboard/assets/charts-6uvCyY0x.js +0 -1
  108. data/vendor/assets/dashboard/assets/i18n-gIeA5VLo.js +0 -1
  109. data/vendor/assets/dashboard/assets/index-BDG9tvBA.css +0 -1
  110. data/vendor/assets/dashboard/assets/index-DWfWAGBc.js +0 -141
  111. data/vendor/assets/dashboard/assets/useJobSetActions-DTDaAoWr.js +0 -1
  112. data/vendor/assets/dashboard/assets/usePageParam-BcAvRko-.js +0 -1
  113. data/vendor/assets/dashboard/assets/useSort-D5Am4bGq.js +0 -1
data/lib/wurk/swarm.rb CHANGED
@@ -5,11 +5,20 @@ require_relative 'launcher'
5
5
  require_relative 'fetcher/reliable'
6
6
  require_relative 'keys'
7
7
  require_relative 'swarm/child_boot'
8
+ require_relative 'swarm/backoff'
9
+ require_relative 'swarm/restart'
10
+ require_relative 'swarm/orphan_guard'
8
11
 
9
12
  module Wurk
10
13
  # Parent supervisor. Forks N children per the worker topology, monitors
11
- # PIDs, relays signals, respawns crashed children, handles rolling
12
- # restart on SIGUSR1, recycles RSS-bloated children.
14
+ # PIDs, relays signals, respawns crashed children with per-slot exponential
15
+ # backoff, handles rolling restart on SIGUSR1, recycles RSS-bloated children.
16
+ #
17
+ # The supervise loop never sleeps on behalf of a respawn or a restart: crash
18
+ # backoff is tracked as per-slot due-times (Swarm::Backoff) and rolling
19
+ # restart / recycle run as a non-blocking state machine (Swarm::Restart)
20
+ # advanced one phase per tick. TERM/INT is therefore honored within a tick
21
+ # regardless of restart or backoff state.
13
22
  #
14
23
  # Boot ordering (must be exact — see docs/idea/03-process-model.md):
15
24
  # 1. Host app boots fully; eager loads done.
@@ -21,11 +30,10 @@ module Wurk
21
30
  # 6. Parent calls `supervise` to enter the wait/relay loop.
22
31
  #
23
32
  # Signals (see docs/idea/04-signals.md):
24
- # TERM/INT → `shutdown` (graceful drain)
25
- # TSTP → relay TSTP (pause fetch)
26
- # CONT → relay CONT (resume fetch)
33
+ # TERM/INT → `shutdown` (graceful drain; aborts any restart)
34
+ # TSTP → relay TSTP (quiet — stop fetching; one-way, no resume)
27
35
  # USR1 → `rolling_restart` (zero-downtime cycle)
28
- class Swarm
36
+ class Swarm # rubocop:disable Metrics/ClassLength
29
37
  include Component
30
38
 
31
39
  SUPERVISE_TICK = 0.2
@@ -34,6 +42,16 @@ module Wurk
34
42
  MEMORY_CHECK_INTERVAL = 10
35
43
  DEFAULT_SHUTDOWN_TIMEOUT = 25
36
44
 
45
+ # Children each hard_shutdown after their own drain deadline (bulk_requeue
46
+ # + a 3s ensure window + heartbeat cleanup); the parent must not SIGKILL
47
+ # them mid-tail, so its own wait always extends past theirs by this much.
48
+ SHUTDOWN_GRACE = 5
49
+
50
+ # USR2 is relayed (log reopen) — without a trap, a logrotate config that
51
+ # signals the master pid would hit USR2's default disposition and kill the
52
+ # whole swarm.
53
+ SWARM_SIGNALS = { 'TERM' => :term, 'INT' => :term, 'TSTP' => :tstp, 'USR1' => :usr1, 'USR2' => :usr2 }.freeze
54
+
37
55
  attr_reader :topology, :children
38
56
 
39
57
  def initialize(topology:, config: Wurk.configuration, memory_limit: config.memory_limit_kb,
@@ -45,28 +63,40 @@ module Wurk
45
63
  @children = {}
46
64
  @assignments = []
47
65
  @stopping = false
66
+ @quieted = false
48
67
  @last_memory_check = 0
49
- @signal_queue = ::Thread::Queue.new
68
+ @signal_read = nil
69
+ @signal_write = nil
70
+ @respawn_backoff = Backoff.new(base: RESPAWN_BACKOFF)
71
+ @restart = build_restart
50
72
  end
51
73
 
52
74
  # `install_signals:` is false in tests so the integration suite can
53
75
  # drive `shutdown` / `rolling_restart` directly without poisoning the
54
76
  # test process's signal handlers.
77
+ #
78
+ # Traps go in BEFORE fork_children: a TERM landing in the (previously
79
+ # post-fork) window between fork and trap installation left the parent on
80
+ # its default disposition — it died instantly and orphaned live, fetching
81
+ # children. Installed first, the trap queues the TERM and the supervise
82
+ # loop drains it (relaying to children) even if it arrives mid-boot.
55
83
  def boot(install_signals: true)
56
84
  raise 'Wurk::Swarm already booted' unless @assignments.empty?
57
85
  raise ArgumentError, 'Topology has no slots' if @topology.empty?
58
86
 
59
87
  @assignments = @topology.assignments.freeze
88
+ install_signal_handlers if install_signals
60
89
  close_parent_sockets
61
90
  fork_children
62
- install_signal_handlers if install_signals
63
91
  @children.keys
64
92
  end
65
93
 
66
94
  def supervise
67
95
  until done?
68
96
  drain_signals
69
- reap_one_child
97
+ reap_children
98
+ spawn_due_respawns
99
+ @restart.advance unless @stopping
70
100
  check_memory_pressure
71
101
  sleep SUPERVISE_TICK
72
102
  end
@@ -74,32 +104,45 @@ module Wurk
74
104
 
75
105
  def shutdown(timeout: @shutdown_timeout)
76
106
  @stopping = true
107
+ @restart.abort
77
108
  relay_signal('TERM')
78
- wait_for_children(timeout)
109
+ wait_for_children(timeout + SHUTDOWN_GRACE)
79
110
  hard_kill_stragglers
80
111
  end
81
112
 
82
- # SIGUSR1. For each existing child, fork a replacement, wait for its
83
- # first heartbeat, then TERM + drain the old one. Long-running jobs
84
- # in the old slot get the full shutdown_timeout while the replacement
85
- # is already serving new work.
113
+ # TSTP quiet is one-way and GLOBAL (spec §21.3): it must survive respawns
114
+ # and memory recycles, or a quieted-but-crashed child's replacement would
115
+ # resume fetching mid-maintenance. The flag makes every future fork boot
116
+ # already-quieted (see ChildBoot start_quiet).
117
+ def quiet_swarm
118
+ @quieted = true
119
+ relay_signal('TSTP')
120
+ end
121
+
122
+ # SIGUSR1: queue every live child for the rolling-restart state machine,
123
+ # which replaces one slot at a time (spawn replacement → await its
124
+ # heartbeat → TERM the old child → await its drain) without blocking the
125
+ # supervise thread, so TERM stays responsive throughout the cycle.
86
126
  def rolling_restart
87
- @children.dup.each do |old_pid, meta|
88
- replacement = fork_child(meta[:slot], meta[:index])
89
- @children[replacement] = meta
90
- unless wait_for_heartbeat(replacement)
91
- logger.warn do
92
- "swarm: replacement #{replacement} heartbeat not seen within #{HEARTBEAT_WAIT}s; proceeding anyway"
93
- end
94
- end
95
- safe_kill(old_pid, 'TERM')
96
- wait_pid(old_pid, @shutdown_timeout)
97
- @children.delete(old_pid)
98
- end
127
+ @restart.enqueue(@children.keys)
99
128
  end
100
129
 
101
130
  private
102
131
 
132
+ def build_restart
133
+ Restart.new(Restart::Config.new(
134
+ spawn: method(:spawn_child),
135
+ kill: method(:safe_kill),
136
+ heartbeat: method(:heartbeat_seen?),
137
+ describe: ->(pid) { @children[pid] },
138
+ now: method(:monotonic),
139
+ logger: logger,
140
+ heartbeat_wait: HEARTBEAT_WAIT,
141
+ drain_timeout: @shutdown_timeout + SHUTDOWN_GRACE,
142
+ backoff: Backoff.new(base: RESPAWN_BACKOFF)
143
+ ))
144
+ end
145
+
103
146
  # Step 3.
104
147
  def close_parent_sockets
105
148
  @config.reset_redis_pools!
@@ -117,82 +160,164 @@ module Wurk
117
160
 
118
161
  # Step 4.
119
162
  def fork_children
120
- @assignments.each_with_index do |slot, idx|
121
- @children[fork_child(slot, idx)] = { slot: slot, index: idx }
122
- end
163
+ @assignments.each_index { |idx| spawn_child(@assignments[idx], idx) }
123
164
  end
124
165
 
166
+ # Fork one child for the slot and record its spawn time so crash backoff can
167
+ # tell a crash-loop (short-lived) from a healthy child that finally died.
168
+ # Returns the child PID; never returns in the child (ChildBoot exits).
169
+ def spawn_child(slot, idx)
170
+ pid = fork_child(slot, idx)
171
+ @children[pid] = { slot: slot, index: idx, spawned_at: monotonic }
172
+ pid
173
+ end
174
+
175
+ # Capture the parent PID BEFORE forking and hand it to the child: read
176
+ # from the parent, it is race-free even if the parent dies the instant
177
+ # after fork (getppid in the child could already return the reaper). The
178
+ # child's OrphanGuard compares live getppid against it.
125
179
  def fork_child(slot, idx)
180
+ parent_pid = ::Process.pid
126
181
  pid = ::Process.fork
127
182
  return pid if pid
128
183
 
129
- ChildBoot.new(@config, slot, idx).run
184
+ # Drop the parent's self-pipe first: the inherited traps still write to
185
+ # it, so a signal landing in the window before ChildBoot resets them
186
+ # would surface in the PARENT's supervise loop (an operator TERMing one
187
+ # child pid would drain the whole swarm). Closed, the trap write no-ops.
188
+ @signal_read&.close
189
+ @signal_write&.close
190
+ ChildBoot.new(@config, slot, idx, parent_pid: parent_pid, start_quiet: @quieted).run
130
191
  exit 0 # unreachable; ChildBoot exits explicitly
131
192
  end
132
193
 
194
+ # Self-pipe pattern (same as Wurk::CLI): the trap only writes the signal
195
+ # name to a pipe — no Thread::Queue#push, which takes a mutex a trap can
196
+ # deadlock against. The supervise loop polls the read end each tick.
133
197
  def install_signal_handlers
134
- { 'TERM' => :term, 'INT' => :term, 'TSTP' => :tstp,
135
- 'CONT' => :cont, 'USR1' => :usr1 }.each do |sig, sym|
136
- ::Signal.trap(sig) { @signal_queue << sym }
198
+ @signal_read, @signal_write = ::IO.pipe
199
+ SWARM_SIGNALS.each_key do |sig|
200
+ ::Signal.trap(sig) { emit_signal(sig) }
201
+ rescue ArgumentError
202
+ # Platform without this signal (e.g. some JRuby builds) — skip it.
203
+ nil
137
204
  end
138
205
  end
139
206
 
207
+ # Non-blocking self-pipe write from trap context: a blocking `puts` could
208
+ # stall signal delivery if the pipe fills. `exception: false` returns
209
+ # :wait_writable instead of raising when full (drop the coalescible
210
+ # duplicate); a closed pipe during shutdown is ignored too.
211
+ def emit_signal(sig)
212
+ @signal_write.write_nonblock("#{sig}\n", exception: false)
213
+ rescue ::IOError, ::Errno::EPIPE, ::Errno::EBADF
214
+ nil
215
+ end
216
+
140
217
  def drain_signals
141
- until @signal_queue.empty?
142
- sym = next_signal_symbol
143
- next if sym.nil?
218
+ return unless @signal_read
144
219
 
145
- case sym
220
+ while (sig = read_pending_signal)
221
+ case SWARM_SIGNALS[sig]
146
222
  when :term then shutdown
147
- when :tstp then relay_signal('TSTP')
148
- when :cont then relay_signal('CONT')
223
+ when :tstp then quiet_swarm
149
224
  when :usr1 then rolling_restart
225
+ when :usr2 then relay_signal('USR2')
150
226
  end
151
227
  end
152
228
  end
153
229
 
154
- def next_signal_symbol
155
- @signal_queue.pop(true)
156
- rescue ThreadError
157
- nil
230
+ # One buffered line per pending signal, non-blocking (wait_readable(0)).
231
+ # nil once the pipe is drained, ending the loop for this tick.
232
+ def read_pending_signal
233
+ return nil unless @signal_read.wait_readable(0)
234
+
235
+ @signal_read.gets&.strip
158
236
  end
159
237
 
160
- def reap_one_child
161
- pid, status = ::Process.wait2(-1, ::Process::WNOHANG)
162
- on_child_exit(pid, status) if pid
238
+ # Reap every exited child this tick (not one), so a fleet-wide death
239
+ # recovers in parallel rather than one child per SUPERVISE_TICK. ECHILD
240
+ # (momentarily no children — all crashed and awaiting backoff) is not a stop
241
+ # condition: the swarm only stops on an explicit TERM/INT.
242
+ def reap_children
243
+ loop do
244
+ pid, status = ::Process.wait2(-1, ::Process::WNOHANG)
245
+ break unless pid
246
+
247
+ on_child_exit(pid, status)
248
+ end
163
249
  rescue Errno::ECHILD
164
- @stopping = true
250
+ nil
165
251
  end
166
252
 
167
253
  def on_child_exit(pid, status)
168
254
  meta = @children.delete(pid)
169
255
  return unless meta
256
+ return if @restart.claim_exit(pid)
170
257
 
171
258
  if @stopping
172
259
  logger.info { "swarm: child #{pid} exited (status=#{status.exitstatus})" }
173
260
  else
174
- logger.warn { "swarm: child #{pid} died (status=#{status.exitstatus}); respawning slot #{meta[:index]}" }
175
- sleep RESPAWN_BACKOFF
176
- @children[fork_child(meta[:slot], meta[:index])] = meta
261
+ schedule_respawn(pid, status, meta)
177
262
  end
178
263
  end
179
264
 
265
+ # Arm the slot's backoff instead of sleeping the supervise thread; the next
266
+ # due tick respawns it. A child that lived past the reset window counts as a
267
+ # fresh failure (base delay), so only a genuine crash-loop escalates toward
268
+ # the cap.
269
+ def schedule_respawn(pid, status, meta)
270
+ idx = meta[:index]
271
+ delay = @respawn_backoff.fail(idx, lifetime: monotonic - meta[:spawned_at])
272
+ logger.warn do
273
+ "swarm: child #{pid} died (status=#{status.exitstatus}); respawning slot #{idx} in #{delay}s"
274
+ end
275
+ end
276
+
277
+ # Respawn any slot whose backoff window has elapsed. Runs every tick; a
278
+ # no-op until a scheduled respawn comes due.
279
+ def spawn_due_respawns
280
+ return if @stopping
281
+
282
+ @assignments.each_index do |idx|
283
+ next unless @respawn_backoff.pending?(idx) && @respawn_backoff.ready?(idx)
284
+
285
+ respawn_slot(idx)
286
+ end
287
+ end
288
+
289
+ # Consume the backoff only after the fork lands. A fork/resource failure
290
+ # (EAGAIN, ENOMEM) must neither drop the pending respawn nor escape the
291
+ # supervise loop: on failure we re-arm the backoff and leave the slot
292
+ # pending so the next due tick retries instead of hot-looping.
293
+ def respawn_slot(idx)
294
+ spawn_child(@assignments[idx], idx)
295
+ @respawn_backoff.consume(idx)
296
+ rescue StandardError => e
297
+ delay = @respawn_backoff.fail(idx)
298
+ logger.warn { "swarm: respawn of slot #{idx} failed (#{e.class}: #{e.message}); retrying in #{delay}s" }
299
+ end
300
+
180
301
  def check_memory_pressure
181
302
  return unless @memory_limit
182
303
 
183
- now = ::Process.clock_gettime(::Process::CLOCK_MONOTONIC)
304
+ now = monotonic
184
305
  return if now - @last_memory_check < MEMORY_CHECK_INTERVAL
185
306
 
186
307
  @last_memory_check = now
187
308
  @children.dup.each_key { |pid| recycle_if_bloated(pid) }
188
309
  end
189
310
 
311
+ # Route a bloated child through the restart state machine (same path as a
312
+ # rolling restart) so recycle is graceful — a healthy replacement takes over
313
+ # before the old child is TERMed — and can't overlap a restart already in
314
+ # flight on the slot.
190
315
  def recycle_if_bloated(pid)
191
316
  rss = pid_rss_kb(pid)
192
317
  return if rss.nil? || rss < @memory_limit
193
318
 
194
319
  logger.warn { "swarm: child #{pid} RSS #{rss}KB >= #{@memory_limit}KB; recycling" }
195
- safe_kill(pid, 'TERM')
320
+ @restart.enqueue([pid])
196
321
  end
197
322
 
198
323
  def pid_rss_kb(pid)
@@ -213,22 +338,10 @@ module Wurk
213
338
  nil
214
339
  end
215
340
 
216
- def wait_pid(pid, timeout)
217
- deadline = ::Process.clock_gettime(::Process::CLOCK_MONOTONIC) + timeout
218
- while ::Process.clock_gettime(::Process::CLOCK_MONOTONIC) < deadline
219
- return true if ::Process.wait(pid, ::Process::WNOHANG)
220
-
221
- sleep 0.1
222
- end
223
- false
224
- rescue Errno::ECHILD
225
- true
226
- end
227
-
228
341
  def wait_for_children(timeout)
229
- deadline = ::Process.clock_gettime(::Process::CLOCK_MONOTONIC) + timeout
230
- while ::Process.clock_gettime(::Process::CLOCK_MONOTONIC) < deadline && @children.any?
231
- reap_one_child
342
+ deadline = monotonic + timeout
343
+ while monotonic < deadline && @children.any?
344
+ reap_children
232
345
  sleep 0.1
233
346
  end
234
347
  end
@@ -238,21 +351,24 @@ module Wurk
238
351
  @children.clear
239
352
  end
240
353
 
241
- # Identity is `<hostname>:<pid>:<nonce>`. PROCESS_NONCE is set when
242
- # Component loads in the parent and inherited by every fork — the
243
- # parent can compute a child's identity from its PID alone.
244
- # Returns true if the heartbeat was observed before the deadline.
245
- def wait_for_heartbeat(pid) # rubocop:disable Naming/PredicateMethod
354
+ # Has the child written its first heartbeat yet? One non-blocking SISMEMBER,
355
+ # polled by the restart state machine each tick. Identity is
356
+ # `<hostname>:<pid>:<nonce>`; PROCESS_NONCE is set when Component loads in
357
+ # the parent and inherited by every fork, so the parent computes a child's
358
+ # identity from its PID alone. A Redis blip returns false (not seen yet) so a
359
+ # transient error can't crash the supervisor — the restart deadline still
360
+ # forces progress.
361
+ def heartbeat_seen?(pid)
246
362
  identity = "#{hostname}:#{pid}:#{Component::PROCESS_NONCE}"
247
- deadline = ::Process.clock_gettime(::Process::CLOCK_MONOTONIC) + HEARTBEAT_WAIT
248
- while ::Process.clock_gettime(::Process::CLOCK_MONOTONIC) < deadline
249
- return true if @config.redis { |c| c.call('SISMEMBER', Keys::PROCESSES, identity) } == 1
250
-
251
- sleep 0.5
252
- end
363
+ @config.redis { |c| c.call('SISMEMBER', Keys::PROCESSES, identity) } == 1
364
+ rescue StandardError
253
365
  false
254
366
  end
255
367
 
368
+ def monotonic
369
+ ::Process.clock_gettime(::Process::CLOCK_MONOTONIC)
370
+ end
371
+
256
372
  def done?
257
373
  @stopping && @children.empty?
258
374
  end
@@ -0,0 +1,49 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Wurk
4
+ # Mutex/CV "tick every N seconds until told to stop" primitive shared by
5
+ # every leader-gated periodic component (History, Metrics::Rollup,
6
+ # Metrics::QueueRollup). Each of those used to hand-roll the same
7
+ # `@mutex`/`@sleeper`/`@done` dance; this is that dance, extracted once.
8
+ #
9
+ # The host owns thread spawning (it needs its own `safe_thread` from
10
+ # Component for logger/handle_exception context) and the leader-gate check
11
+ # inside its `tick` — TimerLoop only owns the interval wait and the
12
+ # start/terminate signaling around it:
13
+ #
14
+ # @timer = Wurk::TimerLoop.new(interval)
15
+ # @thread ||= safe_thread('my-loop') { @timer.run { tick } }
16
+ # def terminate = @timer.terminate
17
+ class TimerLoop
18
+ def initialize(interval)
19
+ @interval = interval
20
+ @done = false
21
+ @mutex = ::Mutex.new
22
+ @sleeper = ::ConditionVariable.new
23
+ end
24
+
25
+ # Waits one interval, then yields repeatedly (waiting between calls)
26
+ # until #terminate is called. Matches the existing components' "don't
27
+ # tick immediately on boot" behavior.
28
+ def run
29
+ wait
30
+ until @done
31
+ yield
32
+ wait
33
+ end
34
+ end
35
+
36
+ def terminate
37
+ @mutex.synchronize do
38
+ @done = true
39
+ @sleeper.signal
40
+ end
41
+ end
42
+
43
+ def wait
44
+ @mutex.synchronize do
45
+ @sleeper.wait(@mutex, @interval) unless @done
46
+ end
47
+ end
48
+ end
49
+ end
data/lib/wurk/version.rb CHANGED
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Wurk
4
- VERSION = "1.1.0"
4
+ VERSION = "1.1.2"
5
5
  end
@@ -271,7 +271,10 @@ module Wurk
271
271
  end
272
272
  rescue ::StandardError => e
273
273
  ::Wurk.configuration.handle_exception(e, context: 'web-extension-render')
274
- [500, html_headers, "Extension render error: #{::CGI.escapeHTML(e.message)}"]
274
+ # Message stays server-side (handle_exception logs it): exception
275
+ # text can carry file paths or Redis connection details, and every
276
+ # dashboard viewer sees this body.
277
+ [500, html_headers, 'Extension render error (see server logs)']
275
278
  end
276
279
 
277
280
  def action_for(ext, route_params, env, ctx)
@@ -0,0 +1,46 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Wurk
4
+ class Web
5
+ # Routes `Wurk.redis` onto the dedicated web pool (Wurk::Configuration
6
+ # #web_redis_pool) for the duration of a block.
7
+ #
8
+ # The dashboard, JSON API, and SSE stream run in the host's web process and
9
+ # reach Redis through the same inspector objects (Stats, Queue, Search, …) a
10
+ # worker uses — all of which call `Wurk.redis`. Left on the default pool, a
11
+ # burst of dashboard traffic or a long-held SSE stream could drain the
12
+ # connections a co-located (embedded) worker needs to make progress, and
13
+ # vice versa: the #101 pool-exhaustion incident.
14
+ #
15
+ # `Wurk.redis_pool` resolves its pool by calling `#redis_pool` on
16
+ # `Thread.current[:wurk_capsule]` (falling back to the default capsule).
17
+ # Pointing that thread-local at a handle whose `#redis_pool` is the web pool
18
+ # diverts every nested `Wurk.redis` call with no per-call-site change. The
19
+ # prior value is restored on exit so a reused web-server thread — or a
20
+ # nested scope — is left untouched.
21
+ module PoolScope
22
+ class << self
23
+ def scope
24
+ prev = Thread.current[:wurk_capsule]
25
+ Thread.current[:wurk_capsule] = handle
26
+ yield
27
+ ensure
28
+ Thread.current[:wurk_capsule] = prev
29
+ end
30
+
31
+ def handle
32
+ @handle ||= Handle.new
33
+ end
34
+ end
35
+
36
+ # Minimal capsule-shaped duck type: Wurk.redis_pool only ever calls
37
+ # `#redis_pool` on the thread-local. Resolved lazily so a post-fork
38
+ # `reset_redis_pools!` rebuild is picked up transparently.
39
+ class Handle
40
+ def redis_pool
41
+ Wurk.configuration.web_redis_pool
42
+ end
43
+ end
44
+ end
45
+ end
46
+ end
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require_relative 'extension'
4
+ require_relative 'pool_scope'
4
5
 
5
6
  module Wurk
6
7
  class Web
@@ -29,7 +30,7 @@ module Wurk
29
30
  # auth here would break `run Sidekiq::Web` / rack-test parity with
30
31
  # upstream Sidekiq, which also doesn't auth-gate `Sidekiq::Web.call`.
31
32
  def call(env)
32
- config.rack_app(method(:dispatch)).call(env)
33
+ PoolScope.scope { config.rack_app(method(:dispatch)).call(env) }
33
34
  end
34
35
 
35
36
  private