wurk 1.3.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +1 -0
  3. data/lib/wurk/batch/callbacks.rb +82 -12
  4. data/lib/wurk/batch/death_handler.rb +7 -4
  5. data/lib/wurk/batch/server_middleware.rb +1 -1
  6. data/lib/wurk/batch.rb +121 -15
  7. data/lib/wurk/capsule.rb +5 -4
  8. data/lib/wurk/cli.rb +48 -14
  9. data/lib/wurk/client/buffered.rb +193 -43
  10. data/lib/wurk/client.rb +87 -14
  11. data/lib/wurk/compat.rb +1 -1
  12. data/lib/wurk/component.rb +2 -2
  13. data/lib/wurk/configuration.rb +10 -2
  14. data/lib/wurk/cron.rb +94 -37
  15. data/lib/wurk/deploy.rb +5 -3
  16. data/lib/wurk/embedded.rb +13 -0
  17. data/lib/wurk/fetcher/reaper.rb +113 -56
  18. data/lib/wurk/fetcher/reliable.rb +62 -9
  19. data/lib/wurk/heartbeat.rb +22 -10
  20. data/lib/wurk/history.rb +13 -1
  21. data/lib/wurk/launcher.rb +133 -66
  22. data/lib/wurk/leader.rb +29 -10
  23. data/lib/wurk/limiter/base.rb +8 -10
  24. data/lib/wurk/limiter/bucket.rb +1 -1
  25. data/lib/wurk/limiter/concurrent.rb +27 -22
  26. data/lib/wurk/limiter/window.rb +13 -11
  27. data/lib/wurk/limiter.rb +7 -4
  28. data/lib/wurk/lua.rb +97 -14
  29. data/lib/wurk/manager.rb +29 -13
  30. data/lib/wurk/metrics/history.rb +4 -3
  31. data/lib/wurk/metrics/queue_rollup.rb +13 -1
  32. data/lib/wurk/metrics/rollup.rb +13 -1
  33. data/lib/wurk/middleware/interrupt_handler.rb +7 -6
  34. data/lib/wurk/middleware/poison_pill.rb +70 -29
  35. data/lib/wurk/middleware.rb +2 -2
  36. data/lib/wurk/pool_checkout.rb +29 -0
  37. data/lib/wurk/process_set.rb +10 -5
  38. data/lib/wurk/processor.rb +6 -0
  39. data/lib/wurk/profiler.rb +3 -2
  40. data/lib/wurk/queue.rb +10 -7
  41. data/lib/wurk/rails_boot.rb +38 -7
  42. data/lib/wurk/redis_client_adapter.rb +48 -4
  43. data/lib/wurk/redis_options.rb +142 -0
  44. data/lib/wurk/redis_pool.rb +102 -39
  45. data/lib/wurk/scheduled.rb +30 -2
  46. data/lib/wurk/stats.rb +14 -9
  47. data/lib/wurk/swarm/child_boot.rb +12 -0
  48. data/lib/wurk/swarm.rb +174 -33
  49. data/lib/wurk/timer_loop.rb +14 -0
  50. data/lib/wurk/version.rb +1 -1
  51. data/lib/wurk/web/enterprise.rb +58 -6
  52. data/lib/wurk/web/extension.rb +1 -1
  53. data/lib/wurk/web/search.rb +5 -3
  54. data/lib/wurk.rb +53 -2
  55. data/vendor/assets/dashboard/assets/{BatchDetail-YRymNsrB.js → BatchDetail-OmC5NPgw.js} +1 -1
  56. data/vendor/assets/dashboard/assets/{Batches-HY4hHdQU.js → Batches-CIpai7St.js} +1 -1
  57. data/vendor/assets/dashboard/assets/{Busy-FCEN1Bpx.js → Busy-A_kwSR6Q.js} +1 -1
  58. data/vendor/assets/dashboard/assets/{Cron-DO3J2zcp.js → Cron-BG7HTqlp.js} +1 -1
  59. data/vendor/assets/dashboard/assets/{Dashboard-B9rOrkzk.js → Dashboard-A_ToqHoo.js} +1 -1
  60. data/vendor/assets/dashboard/assets/{Dead-Bi4GGk9a.js → Dead-8J21jMyK.js} +1 -1
  61. data/vendor/assets/dashboard/assets/Extension-B4Q9FIQu.js +1 -0
  62. data/vendor/assets/dashboard/assets/{FilterBox-IJkHYpdm.js → FilterBox-Fh_Ae7UW.js} +1 -1
  63. data/vendor/assets/dashboard/assets/{JobDetailModal-DS1ypyoc.js → JobDetailModal-Ceng0PMB.js} +1 -1
  64. data/vendor/assets/dashboard/assets/{Limiters-Nz7UbNeJ.js → Limiters-CruDWvNZ.js} +1 -1
  65. data/vendor/assets/dashboard/assets/{Metrics-BBTDxcaE.js → Metrics-CIT7VCoN.js} +1 -1
  66. data/vendor/assets/dashboard/assets/Modal-CN3rdKA_.js +1 -0
  67. data/vendor/assets/dashboard/assets/{Queues-D9PH_THs.js → Queues-D86FYohJ.js} +1 -1
  68. data/vendor/assets/dashboard/assets/{Retries-CAKzDgYG.js → Retries-Bz1O1D-i.js} +1 -1
  69. data/vendor/assets/dashboard/assets/{Scheduled-DTYw1X8S.js → Scheduled-B6h2akTu.js} +1 -1
  70. data/vendor/assets/dashboard/assets/{Search-c4vFxDG_.js → Search-OOu22e5s.js} +1 -1
  71. data/vendor/assets/dashboard/assets/index-BdiUEDXX.css +1 -0
  72. data/vendor/assets/dashboard/assets/index-D_lSDwKw.js +141 -0
  73. data/vendor/assets/dashboard/assets/{useResetPageOnEmpty-B_FsMah6.js → useResetPageOnEmpty-dVPGEWzn.js} +1 -1
  74. data/vendor/assets/dashboard/index.html +2 -2
  75. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  76. metadata +22 -20
  77. data/vendor/assets/dashboard/assets/Extension-BSv8ddW_.js +0 -1
  78. data/vendor/assets/dashboard/assets/Modal-Crrsu64-.js +0 -1
  79. data/vendor/assets/dashboard/assets/index-BxjdeuOa.css +0 -1
  80. data/vendor/assets/dashboard/assets/index-DQu7WY9y.js +0 -141
@@ -11,14 +11,17 @@ module Wurk
11
11
  # Wurk::Launcher so the launcher can stay focused on lifecycle and so the
12
12
  # heartbeat schema lives in one place readers can grep for.
13
13
  #
14
- # Each beat is one pipelined round-trip:
14
+ # Each beat is two pipelined round-trips. The identity write:
15
15
  # SADD processes <identity>
16
16
  # HSET <identity> info concurrency busy beat quiet rss rtt_us
17
17
  # EXPIRE <identity> 60
18
18
  # UNLINK <identity>:work
19
19
  # HSET <identity>:work <tid> <json> ... (only if WORK_STATE non-empty)
20
20
  # EXPIRE <identity>:work 60 (only if WORK_STATE non-empty)
21
+ # then the signal drain:
21
22
  # LPOP <identity>-signals × BEAT_PAUSE
23
+ # Two rather than one because only the first is safe to replay — see
24
+ # #pipelined_beat and #drain_signals.
22
25
  #
23
26
  # The work hash is UNLINK-then-rewritten on every beat — a dropped beat
24
27
  # momentarily empties it, and ProcessSet#cleanup compensates by SREM-ing
@@ -90,16 +93,17 @@ module Wurk
90
93
 
91
94
  private
92
95
 
93
- # Two extra writes (HSET + EXPIRE for the work mirror) when WORK_STATE
94
- # has entries, so `lead` shifts to skip them when slicing signals out
95
- # of the pipeline result.
96
+ # Every command in #write_beat converges on the state this snapshot
97
+ # describes however many times it lands, so the beat claims apply-safety and
98
+ # rides out a blip the way it did before the pool started splitting replay
99
+ # by it. `rtt_us` measures this write alone — the signal drain that follows
100
+ # is a separate checkout and deliberately not part of the gauge.
96
101
  def pipelined_beat
97
102
  work_snapshot = Processor::WORK_STATE.dup
98
- lead = 4 + (work_snapshot.empty? ? 0 : 2)
99
103
  t0 = ::Process.clock_gettime(::Process::CLOCK_MONOTONIC, :microsecond)
100
- results = redis { |conn| conn.pipelined { |pipe| write_beat(pipe, work_snapshot) } }
104
+ redis(idempotent: true) { |conn| conn.pipelined { |pipe| write_beat(pipe, work_snapshot) } }
101
105
  rtt = ::Process.clock_gettime(::Process::CLOCK_MONOTONIC, :microsecond) - t0
102
- [results[lead, BEAT_PAUSE] || [], rtt]
106
+ [drain_signals, rtt]
103
107
  end
104
108
 
105
109
  def write_beat(pipe, work_snapshot)
@@ -107,7 +111,6 @@ module Wurk
107
111
  pipe.call('HSET', @identity, *beat_hash_args(work_snapshot.size))
108
112
  pipe.call('EXPIRE', @identity, TTL_SECONDS)
109
113
  write_work_hash(pipe, work_snapshot)
110
- drain_signals(pipe)
111
114
  end
112
115
 
113
116
  def beat_hash_args(busy)
@@ -132,10 +135,19 @@ module Wurk
132
135
  pipe.call('EXPIRE', work_key, TTL_SECONDS)
133
136
  end
134
137
 
138
+ # Its own checkout, and never an idempotent one: LPOP is destructive and
139
+ # carries its result in the reply, so a replay after a lost reply discards
140
+ # whatever the first attempt already popped — and a discarded entry is a
141
+ # dashboard TERM or TSTP this process never acts on. Fused into the beat
142
+ # pipeline it would have dragged those writes down to the same no-replay
143
+ # default, or worse, invited a later sweep to claim the LPOPs alongside
144
+ # them. Runs after the beat so a signal only leaves Redis once the write
145
+ # that reports us alive has landed.
146
+ #
135
147
  # LPOP one entry per second of cadence so a flood of queued signals
136
148
  # can't stall the beat; anything older drains on the next beat.
137
- def drain_signals(pipe)
138
- BEAT_PAUSE.times { pipe.call('LPOP', "#{@identity}-signals") }
149
+ def drain_signals
150
+ redis { |conn| conn.pipelined { |pipe| BEAT_PAUSE.times { pipe.call('LPOP', "#{@identity}-signals") } } }
139
151
  end
140
152
 
141
153
  def info_hash
data/lib/wurk/history.rb CHANGED
@@ -71,11 +71,23 @@ module Wurk
71
71
  end
72
72
 
73
73
  def start
74
- @thread ||= safe_thread('history-snapshot') { @timer.run { tick } } # rubocop:disable Naming/MemoizedInstanceVariableName
74
+ return @thread if @thread
75
+
76
+ @timer.reset
77
+ @thread = safe_thread('history-snapshot') { @timer.run { tick } }
75
78
  end
76
79
 
80
+ # Blocks until the thread is really gone: the launcher releases the cluster
81
+ # lock immediately after this returns, and a snapshot still in flight would
82
+ # race the next leader's first one.
83
+ #
84
+ # Cleared only on a confirmed join (Thread#join returns nil on timeout): a
85
+ # wedged thread must stay tracked so #start's guard returns it instead of
86
+ # calling @timer.reset, which would un-terminate the loop it is still
87
+ # inside and leave two threads writing the same stream.
77
88
  def terminate
78
89
  @timer.terminate
90
+ @thread = nil if @thread&.join(TimerLoop::JOIN_TIMEOUT)
79
91
  end
80
92
 
81
93
  # Leader-gated: only the elected leader emits, so N workers don't each
data/lib/wurk/launcher.rb CHANGED
@@ -13,6 +13,7 @@ require_relative 'metrics/rollup'
13
13
  require_relative 'metrics/queue_rollup'
14
14
  require_relative 'history'
15
15
  require_relative 'fetcher/reaper'
16
+ require_relative 'timer_loop'
16
17
 
17
18
  module Wurk
18
19
  # Top-level supervisor inside each worker process. Owns the Manager pool
@@ -43,18 +44,23 @@ module Wurk
43
44
  # (Sidekiq's drop-in surface). The single source of truth is Heartbeat.
44
45
  BEAT_PAUSE = Heartbeat::BEAT_PAUSE
45
46
 
47
+ # Bound on how long #stop waits for the boot-time reclaim sweep before
48
+ # moving on — it can still be scanning a large keyspace when a fast
49
+ # shutdown lands right after boot; teardown must not hang on it.
50
+ BOOT_RECLAIM_JOIN_TIMEOUT = 5
51
+
46
52
  attr_accessor :managers, :poller, :cron_poller, :metrics_rollup, :queue_rollup, :history
47
53
 
48
54
  def initialize(config, embedded: false)
49
55
  @config = config
50
56
  @embedded = embedded
51
- # Two separate flags, deliberately. @done = "quieted" (stop fetching, stay
52
- # alive, report quiet=true). @stopped = "shutting down" (terminate the
53
- # heartbeat loop). Quiet must NOT stop the heartbeat — otherwise a quieted
54
- # process never publishes quiet=true and expires out of the live set (#236).
57
+ # @done is "quieted": stop fetching, stay alive, report quiet=true. It
58
+ # deliberately does NOT stop the heartbeat — a quieted process that stopped
59
+ # beating would never publish quiet=true and would expire out of the live
60
+ # set (#236). Only #stop ends the beat, by terminating @beat_timer.
55
61
  @done = false
56
- @stopped = false
57
- @managers = config.capsules.values.map { |cap| Manager.new(cap) }
62
+ @beat_timer = TimerLoop.new(BEAT_PAUSE)
63
+ @managers = build_managers
58
64
  @poller = build_poller
59
65
  @cron_poller = build_cron_poller
60
66
  @metrics_rollup = build_metrics_rollup
@@ -87,18 +93,21 @@ module Wurk
87
93
  @config.capsules.each_value(&:prepare!)
88
94
  @config.freeze!
89
95
  @heartbeat_thread = safe_thread('heartbeat', &method(:start_heartbeat)) if async_beat
90
- @poller&.start
91
- @leader&.start
92
- @cron_poller&.start
93
- @metrics_rollup&.start
94
- @queue_rollup&.start
95
- @history&.start
96
+ [@poller, @leader, @cron_poller, @metrics_rollup, @queue_rollup, @history].compact.each(&:start)
96
97
  @managers.each(&:start)
97
98
  @reaper.start
98
99
  # Run on a background thread so /ready probe isn't delayed by a large
99
100
  # orphan sweep (reaper.reclaim! is atomic, but can scan many entries).
100
101
  @boot_reclaim_thread = safe_thread('boot-reclaim', &method(:boot_reclaim))
101
102
  @health_server&.start
103
+ rescue StandardError
104
+ # Boot is not atomic: whatever raised (a health-check port already bound,
105
+ # ThreadError at the OS thread limit) leaves the steps before it holding
106
+ # threads, sockets and a leader campaign — and the caller is about to drop
107
+ # its only reference to us, so nothing else can ever release them. Guarded,
108
+ # because the caller must see the boot failure, not a rollback failure.
109
+ teardown_step('boot-rollback') { stop }
110
+ raise
102
111
  end
103
112
 
104
113
  # Idempotent. Flips `stopping?` true, halts fetching across every
@@ -118,26 +127,18 @@ module Wurk
118
127
 
119
128
  # Graceful shutdown. Deadline is monotonic so wall-clock skew can't
120
129
  # extend it. Managers stop in parallel threads so a slow capsule
121
- # doesn't block its siblings.
130
+ # doesn't block its siblings — and each drain is guarded, because a
131
+ # capsule that blows up mid-drain (Redis down during bulk_requeue) used
132
+ # to surface out of `join` and skip the whole teardown tail, leaving the
133
+ # leader lock held, the process listed as live, and its port open.
122
134
  def stop
123
135
  deadline = ::Process.clock_gettime(::Process::CLOCK_MONOTONIC) + (@config[:timeout] || 25)
124
136
  quiet
125
- stoppers = @managers.map { |m| Thread.new { m.stop(deadline) } }
137
+ stoppers = @managers.map { |m| Thread.new { teardown_step('manager') { m.stop(deadline) } } }
126
138
  fire_event(:shutdown, reverse: true)
127
139
  stoppers.each(&:join)
128
- # Full shutdown stops periodic firing (it survived #quiet); do this before
129
- # releasing the lock so no tick races a follower's promotion.
130
- @cron_poller&.terminate
131
- @metrics_rollup&.terminate
132
- @queue_rollup&.terminate
133
- @history&.terminate
134
- @reaper&.stop
135
- # CAS-release the cluster lock now (planned shutdown) so a follower can
136
- # take over immediately instead of waiting out the TTL.
137
- @leader&.stop
138
- stop_heartbeat
139
- clear_heartbeat
140
- fire_event(:exit, reverse: true)
140
+ ensure
141
+ release_components
141
142
  end
142
143
 
143
144
  def stopping?
@@ -162,9 +163,12 @@ module Wurk
162
163
 
163
164
  write_stats(processed, failed, expired)
164
165
  rescue StandardError => e
165
- # Replay-safety: counters were reset above, so a Redis blip would
166
- # otherwise drop stats. We log and accept — the per-job at-least-once
167
- # semantics don't apply to *counters*, and the next beat resets again.
166
+ # The counters were reset above, so a raise here drops this window's stats
167
+ # for good: #write_stats cannot claim apply-safety and the pool no longer
168
+ # replays it through a blip. Accepted deliberately — adding them back
169
+ # would double-count the case where the INCRBYs did land and only the
170
+ # reply was lost, and the per-job at-least-once semantics don't apply to
171
+ # *counters*. The next beat resets again.
168
172
  handle_exception(e, { context: 'flush_stats' })
169
173
  end
170
174
 
@@ -174,6 +178,60 @@ module Wurk
174
178
 
175
179
  private
176
180
 
181
+ # Teardown tail, driven from #stop's ensure. Every release is guarded on
182
+ # its own because they are independent: a Redis blip in one (leader CAS
183
+ # release, heartbeat clear) must not strand the ones after it — the process
184
+ # is exiting either way, so the only thing worse than a failed release is a
185
+ # skipped one. Order matters twice over: full shutdown stops the periodic
186
+ # loops (they survived #quiet) before the cluster lock is CAS-released, so
187
+ # no tick races a follower's promotion, and the release itself happens on a
188
+ # planned shutdown rather than waiting out the lock TTL.
189
+ def release_components
190
+ # Joined first, before anything below tears down the Redis pool it's
191
+ # still reading from: it started as a fire-and-forget thread in #run
192
+ # and a shutdown landing right after boot could otherwise race a
193
+ # pool disconnect mid-scan. Bounded: a full-keyspace scan can outlast
194
+ # the timeout, in which case the thread is left running rather than
195
+ # blocking shutdown on it — #boot_reclaim already rescues and logs on
196
+ # its own, so a straggler racing #reset_redis_pools! below just means
197
+ # a noisy log line, not a hang.
198
+ teardown_step('boot-reclaim-join') { @boot_reclaim_thread&.join(BOOT_RECLAIM_JOIN_TIMEOUT) }
199
+ stop_periodic_components
200
+ %i[stop_heartbeat clear_heartbeat].each { |step| teardown_step(step) { send(step) } }
201
+ # Unguarded: fire_event already reports and skips past a raising hook.
202
+ fire_event(:exit, reverse: true)
203
+ # Embedded only: a swarm child or standalone process exits right after
204
+ # #stop anyway, so disconnecting here would just make every unit test
205
+ # that inspects Redis post-stop rebuild a pool for no reason. Embedded
206
+ # hosts (Puma, a rake task) keep running and can `run` again later —
207
+ # without this a stop-then-run cycle doubles the live socket set.
208
+ teardown_step('redis-pools') { @config.reset_redis_pools! } if @embedded
209
+ ensure
210
+ # In an ensure of its own, not merely a guarded step: a leaked TCPServer
211
+ # FD survives even a non-StandardError unwind (a second TERM landing
212
+ # mid-teardown), and kubelet would keep getting 200s from a process that
213
+ # is already gone.
214
+ teardown_step('health-server') { @health_server&.stop }
215
+ end
216
+
217
+ # Split out of #release_components to keep it under the AbcSize/
218
+ # CyclomaticComplexity ceilings — these three are the "periodic loop"
219
+ # releases, independently guarded like everything else in the tail.
220
+ def stop_periodic_components
221
+ [@cron_poller, @metrics_rollup, @queue_rollup, @history].each { |t| teardown_step(t.class) { t&.terminate } }
222
+ teardown_step('reaper') { @reaper&.stop }
223
+ teardown_step('leader') { @leader&.stop }
224
+ end
225
+
226
+ def teardown_step(label)
227
+ yield
228
+ rescue StandardError => e
229
+ handle_exception(e, { context: "launcher-stop-#{label}" })
230
+ end
231
+
232
+ # INCRBY is additive, so a pipeline replayed after a lost reply counts this
233
+ # window twice on every `stat:` key it touches — no apply-safety claim. (The
234
+ # EXPIREs riding along are harmless to repeat; the INCRBYs pin the block.)
177
235
  def write_stats(processed, failed, expired)
178
236
  day = Time.now.utc.strftime('%F')
179
237
  @config.redis do |conn|
@@ -217,74 +275,83 @@ module Wurk
217
275
 
218
276
  # Erase the live-process footprint. flush_stats first so we don't drop
219
277
  # the final batch of counters; then Heartbeat#stop! removes us from the
220
- # `processes` SET and UNLINK-s the identity + work hashes. The probe
221
- # server is closed alongside so kubelet stops getting 200s after the
222
- # process is no longer healthy.
278
+ # `processes` SET and UNLINK-s the identity + work hashes.
223
279
  def clear_heartbeat
224
280
  flush_stats
225
281
  @heartbeat&.stop!
226
- @health_server&.stop
227
282
  end
228
283
 
229
284
  # Terminate the heartbeat loop and wait for it to exit before clear_heartbeat
230
285
  # removes us from the `processes` SET — otherwise a final in-flight beat could
231
- # SADD us back right after the SREM. Wakes the thread out of its BEAT_PAUSE
232
- # sleep so shutdown isn't delayed up to a full interval.
286
+ # SADD us back right after the SREM and the identity would linger for a full
287
+ # 60s TTL.
288
+ #
289
+ # The join is unbounded on purpose. This used to be `wakeup` + `join(BEAT_PAUSE)`,
290
+ # which lost the race whenever the beat was mid-Redis-call: `wakeup` does nothing
291
+ # to a thread that isn't sleeping, so the loop then slept a full interval and the
292
+ # bounded join returned with the thread still live — exactly the resurrect-after-
293
+ # SREM this ordering exists to prevent. A condvar can't be missed (terminate flips
294
+ # the flag inside the critical section the loop re-checks it in), so all that is
295
+ # left to wait out is one in-flight beat, whose Redis calls are timeout-bounded.
233
296
  def stop_heartbeat
234
- @stopped = true
297
+ @beat_timer.terminate
235
298
  thread = @heartbeat_thread
236
299
  return unless thread
237
- # Embedded dashboard-TERM runs `stop` from the beat itself; a self-join
238
- # raises ThreadError. @stopped is set, so the loop exits after this beat.
300
+ # Embedded dashboard-TERM can run `stop` from the beat itself; a self-join
301
+ # raises ThreadError. The timer is terminated, so the loop exits after this beat.
239
302
  return if thread == Thread.current
240
303
 
241
- begin
242
- thread.wakeup
243
- rescue ThreadError
244
- nil
245
- end
246
- thread.join(BEAT_PAUSE)
304
+ thread.join
247
305
  end
248
306
 
249
- # Heartbeat thread loop. `safe_thread` already wraps exceptions. Loops on
250
- # `@stopped` — NOT `@done` — so a *quieted* process keeps beating and publishes
251
- # `quiet=true` instead of vanishing from the live set (#236). Only `#stop`
252
- # flips `@stopped`; its `Thread#wakeup` breaks the sleep so the loop re-checks
253
- # `@stopped` and exits without waiting out the interval.
307
+ # Heartbeat thread loop. `safe_thread` already wraps exceptions. Beats once up
308
+ # front — TimerLoop#run waits before its first yield, and the dashboard has to
309
+ # see this process the moment it can pick up jobs — then ticks until #stop
310
+ # terminates the timer. Note it does NOT stop on `@done`: a *quieted* process
311
+ # keeps beating so it publishes `quiet=true` instead of vanishing (#236).
254
312
  def start_heartbeat
255
- until @stopped
256
- heartbeat
257
- sleep BEAT_PAUSE
258
- end
313
+ heartbeat
314
+ @beat_timer.run { heartbeat }
259
315
  logger.info('Heartbeat stopping...')
260
316
  end
261
317
 
262
318
  # Dashboard-queued signals must behave exactly like OS signals, so a
263
- # standalone process re-delivers to itself and lets the installed trap
264
- # run — that wakes the main thread (CLI self-pipe / child dispatcher)
265
- # so the process actually exits instead of stopping its managers and
266
- # then parking forever. Embedded mode owns no traps (and self-TERM
267
- # would kill the host app), so it calls quiet/stop directly — stop on
268
- # its own thread because `stop` joins the heartbeat thread we're on.
319
+ # standalone process re-delivers to itself and lets the installed trap run —
320
+ # that wakes the main thread (CLI self-pipe / child dispatcher) so the
321
+ # process actually exits instead of stopping its managers and then parking
322
+ # forever. Embedded mode owns no traps (and self-TERM would kill the host
323
+ # app), so quiet applies directly.
269
324
  def dispatch_signal(sig)
270
325
  case sig
271
- when 'TSTP', 'TERM'
272
- if @embedded
273
- sig == 'TSTP' ? quiet : Thread.new { stop }
274
- else
275
- redeliver(sig)
276
- end
326
+ when 'TSTP' then @embedded ? quiet : redeliver(sig)
327
+ when 'TERM' then request_shutdown
277
328
  else
278
329
  logger.warn { "Unknown signal in #{identity}-signals: #{sig.inspect}" }
279
330
  end
280
331
  end
281
332
 
333
+ # The one way anything inside this process asks it to shut down gracefully:
334
+ # dashboard-queued TERM, or a Manager that can no longer hold its
335
+ # concurrency. Standalone hands off to the installed TERM trap; embedded
336
+ # owns no traps, so it drains in place — on its own thread, because callers
337
+ # may be a thread `stop` itself joins (the heartbeat) or kills (a Processor).
338
+ def request_shutdown
339
+ @embedded ? Thread.new { stop } : redeliver('TERM')
340
+ end
341
+
282
342
  # Separate method so tests can stub it — really sending TERM/TSTP would
283
343
  # kill or suspend the test process.
284
344
  def redeliver(sig)
285
345
  ::Process.kill(sig, ::Process.pid)
286
346
  end
287
347
 
348
+ # One Manager per capsule, each holding our shutdown request: a Manager
349
+ # that can no longer replace a dead Processor has to take the process
350
+ # down, and only the Launcher knows how this process exits.
351
+ def build_managers
352
+ @config.capsules.values.map { |cap| Manager.new(cap, shutdown: method(:request_shutdown)) }
353
+ end
354
+
288
355
  def build_poller
289
356
  Wurk::Scheduled::Poller.new(@config)
290
357
  end
data/lib/wurk/leader.rb CHANGED
@@ -3,6 +3,8 @@
3
3
  require 'securerandom'
4
4
  require 'socket'
5
5
  require_relative 'component'
6
+ require_relative 'lua'
7
+ require_relative 'pool_checkout'
6
8
 
7
9
  module Wurk
8
10
  # Cluster leader election via Redis `SET NX EX`. Single-leader-per-cluster
@@ -89,9 +91,19 @@ module Wurk
89
91
 
90
92
  # CAS DEL — only drop the key if we still own it, otherwise a stale
91
93
  # release would yank leadership from whichever follower took over.
94
+ # GET-then-DEL is not that CAS: our key can lapse between the two
95
+ # commands and a follower can win the election in the gap, whereupon
96
+ # the bare DEL deletes *its* lock and leaves the cluster leaderless
97
+ # (or, once the ex-leader re-campaigns, doubly led) for up to one
98
+ # renew interval — double cron fires, double rollups. The compare and
99
+ # the delete happen in one server-side step instead, via the script
100
+ # `Unique.release_if_owner` already shares.
101
+ #
102
+ # `idempotent:` — a replay after a lost reply finds the key either
103
+ # already gone or no longer ours, so it can only be a no-op.
92
104
  def release
93
- redis_call do |c|
94
- c.call('DEL', @key) if c.call('GET', @key) == @owner
105
+ redis_call(idempotent: true) do |c|
106
+ Wurk::Lua::Loader.eval_cached(c, :release_if_owner, keys: [@key], argv: [@owner])
95
107
  end
96
108
  @held = false
97
109
  @token = nil
@@ -107,6 +119,11 @@ module Wurk
107
119
  # follower, it polls every `follower_interval`. Caller must invoke
108
120
  # `stop` for orderly shutdown — the thread also releases its lock on
109
121
  # exit.
122
+ #
123
+ # The spawn happens *inside* the lock: with it outside, two callers
124
+ # racing into `start` both clear the nil check and both spawn, and the
125
+ # loser's loop is never recorded — an unjoinable second campaigner that
126
+ # outlives `stop` and keeps renewing the cluster lock.
110
127
  def start
111
128
  return nil if disabled?
112
129
 
@@ -114,17 +131,19 @@ module Wurk
114
131
  return @thread if @thread
115
132
 
116
133
  @done = false
134
+ @thread = spawn_loop_thread
117
135
  end
118
- @thread = spawn_loop_thread
136
+ @thread
119
137
  end
120
138
 
121
139
  def stop
122
- @mutex.synchronize do
140
+ thread = @mutex.synchronize do
123
141
  @done = true
124
142
  @sleeper.signal
143
+ @thread
125
144
  end
126
- @thread&.join
127
- @thread = nil
145
+ thread&.join
146
+ @mutex.synchronize { @thread = nil }
128
147
  release
129
148
  end
130
149
 
@@ -175,13 +194,13 @@ module Wurk
175
194
  @config.handle_exception(e, event: :leader) if @config.respond_to?(:handle_exception)
176
195
  end
177
196
 
178
- def redis_call(&)
197
+ def redis_call(idempotent: false, &)
179
198
  if @pool
180
- @pool.with(&)
199
+ PoolCheckout.with(@pool, idempotent, &)
181
200
  elsif @config
182
- @config.redis(&)
201
+ @config.redis(idempotent:, &)
183
202
  else
184
- Wurk.redis(&)
203
+ Wurk.redis(idempotent:, &)
185
204
  end
186
205
  end
187
206
 
@@ -104,17 +104,15 @@ module Wurk
104
104
  end
105
105
  end
106
106
 
107
+ # Metadata HSET + EXPIRE + `lmtr-list` SADD in a single round trip. The
108
+ # script also gates the SADD on the metadata being newly written, so a
109
+ # per-job construction (`stripe-#{user_id}`) stops rewriting a set that
110
+ # grows one member per name — see limiter_register.lua for why that gate
111
+ # is only safe while it is atomic with the HSET.
107
112
  def register!
108
- Wurk::Limiter.redis do |c|
109
- c.call('SADD', LIST_KEY, @name)
110
- c.call(
111
- 'HSET', meta_key,
112
- 'type', type.to_s,
113
- 'options', JSON.dump(serializable_options),
114
- 'fingerprint', fingerprint
115
- )
116
- c.call('EXPIRE', meta_key, @options[:ttl])
117
- end
113
+ lua(:limiter_register,
114
+ keys: [meta_key, LIST_KEY],
115
+ argv: [@name, type.to_s, JSON.dump(serializable_options), fingerprint, ttl])
118
116
  end
119
117
 
120
118
  def ttl
@@ -38,7 +38,7 @@ module Wurk
38
38
  remaining = deadline - ::Time.now.to_f
39
39
  raise OverLimit, self if remaining <= 0
40
40
 
41
- sleep [remaining, secs_to_next.to_f, 0.05].compact.min.clamp(0.0, remaining)
41
+ sleep [0.05, secs_to_next.to_f].max.clamp(0.0, remaining)
42
42
  end
43
43
  end
44
44
 
@@ -6,7 +6,7 @@ module Wurk
6
6
  module Limiter
7
7
  # Atomic slot acquisition in a ZSET. Score = expiry epoch; the acquire
8
8
  # script first evicts expired slots (bumping the `reclaimed` metric)
9
- # then ZADDs if there's headroom.
9
+ # then ZADDs if there's headroom (bumping the `held` metric).
10
10
  #
11
11
  # On exhaustion: spin loop with backoff. The spec says "blocks via
12
12
  # Redis stream XREAD" — that's a perf optimization; the visible
@@ -38,33 +38,15 @@ module Wurk
38
38
  raise ArgumentError, 'block required' unless block
39
39
 
40
40
  started = monotime
41
- deadline = started + @options[:wait_timeout]
42
41
  slot = random_id
43
- acquired_at = nil
44
- loop do
45
- result = acquire(slot)
46
- if result[0].to_i == 1
47
- acquired_at = monotime
48
- break
49
- end
50
-
51
- return if @options[:policy] == :ignore
52
-
53
- remaining = deadline - monotime
54
- if remaining <= 0
55
- bump_counter('overages')
56
- raise OverLimit, self
57
- end
58
-
59
- sleep [remaining, WAIT_SLEEP].min
60
- end
42
+ acquired_at = wait_for_slot(slot, started + @options[:wait_timeout])
43
+ return unless acquired_at
61
44
 
62
45
  begin
63
46
  incr_immediate_or_waited(acquired_at - started)
64
47
  block.call
65
48
  ensure
66
- release(slot)
67
- bump_counter('held_time', (monotime - acquired_at).to_i) if acquired_at
49
+ record_release(slot, acquired_at)
68
50
  end
69
51
  end
70
52
 
@@ -98,6 +80,29 @@ module Wurk
98
80
  "lmtr-stats:#{@name}"
99
81
  end
100
82
 
83
+ # A ZREM that removes nothing means an acquirer already evicted our slot:
84
+ # we outran `lock_timeout`, which is what `overages` counts. Exhausting
85
+ # `wait_timeout` is a different event — it raises OverLimit and the server
86
+ # middleware counts it in the job's `overrated` field.
87
+ def record_release(slot, acquired_at)
88
+ bump_counter('overages') if release(slot).to_i.zero?
89
+ bump_counter('held_time', (monotime - acquired_at).to_i)
90
+ end
91
+
92
+ # Spins until a slot frees up. Returns the monotonic acquire time, nil
93
+ # when `policy: :ignore` gives up; raises OverLimit past `wait_timeout`.
94
+ def wait_for_slot(slot, deadline)
95
+ loop do
96
+ return monotime if acquire(slot)[0].to_i == 1
97
+ return if @options[:policy] == :ignore
98
+
99
+ remaining = deadline - monotime
100
+ raise OverLimit, self if remaining <= 0
101
+
102
+ sleep [remaining, WAIT_SLEEP].min
103
+ end
104
+ end
105
+
101
106
  def acquire(slot)
102
107
  lua(:limiter_concurrent_acquire,
103
108
  keys: [state_key, stats_key],
@@ -19,18 +19,15 @@ module Wurk
19
19
  end
20
20
 
21
21
  def size
22
- cutoff = ::Time.now.to_f - interval_seconds
23
- Wurk::Limiter.redis do |c|
24
- c.call('ZREMRANGEBYSCORE', state_key, '-inf', "(#{cutoff}")
25
- c.call('ZCARD', state_key).to_i
26
- end
22
+ window_state.first
27
23
  end
28
24
 
29
25
  # used = entries still inside the window; limit = count; reset_at =
30
26
  # when the oldest entry slides out (freeing a slot), or nil when
31
27
  # idle (#16).
32
28
  def status
33
- build_status(used: size, limit: @options[:count], reset_at: oldest_expiry)
29
+ used, oldest = window_state
30
+ build_status(used: used, limit: @options[:count], reset_at: oldest && (oldest + interval_seconds))
34
31
  end
35
32
 
36
33
  def within_limit(used: 1, &block)
@@ -60,11 +57,16 @@ module Wurk
60
57
  "lmtr-w:#{@name}"
61
58
  end
62
59
 
63
- # Oldest timestamp + interval = the moment it leaves the window.
64
- def oldest_expiry
65
- row = Wurk::Limiter.redis { |c| c.call('ZRANGE', state_key, 0, 0, 'WITHSCORES') }
66
- score = Wurk::Limiter.first_score(row)
67
- score && (score + interval_seconds)
60
+ # Count + oldest in-window score, both scoped by the Redis clock, in one
61
+ # read-only round trip. Reads never trim (the acquire script owns that):
62
+ # a dashboard host whose clock ran ahead used to evict live entries just
63
+ # by rendering `status`, freeing the window for a second full charge.
64
+ # Oldest timestamp + interval = the moment it leaves the window; -1 =
65
+ # nothing in window.
66
+ def window_state
67
+ count, oldest = lua(:limiter_window_status, keys: [state_key], argv: [interval_seconds])
68
+ score = oldest.to_f
69
+ [count.to_i, score.negative? ? nil : score]
68
70
  end
69
71
 
70
72
  def interval_seconds