wurk 1.3.0 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +1 -0
- data/lib/wurk/batch/callbacks.rb +82 -12
- data/lib/wurk/batch/death_handler.rb +7 -4
- data/lib/wurk/batch/server_middleware.rb +1 -1
- data/lib/wurk/batch.rb +121 -15
- data/lib/wurk/capsule.rb +5 -4
- data/lib/wurk/cli.rb +48 -14
- data/lib/wurk/client/buffered.rb +193 -43
- data/lib/wurk/client.rb +87 -14
- data/lib/wurk/compat.rb +1 -1
- data/lib/wurk/component.rb +2 -2
- data/lib/wurk/configuration.rb +10 -2
- data/lib/wurk/cron.rb +94 -37
- data/lib/wurk/deploy.rb +5 -3
- data/lib/wurk/embedded.rb +13 -0
- data/lib/wurk/fetcher/reaper.rb +113 -56
- data/lib/wurk/fetcher/reliable.rb +62 -9
- data/lib/wurk/heartbeat.rb +22 -10
- data/lib/wurk/history.rb +13 -1
- data/lib/wurk/launcher.rb +133 -66
- data/lib/wurk/leader.rb +29 -10
- data/lib/wurk/limiter/base.rb +8 -10
- data/lib/wurk/limiter/bucket.rb +1 -1
- data/lib/wurk/limiter/concurrent.rb +27 -22
- data/lib/wurk/limiter/window.rb +13 -11
- data/lib/wurk/limiter.rb +7 -4
- data/lib/wurk/lua.rb +97 -14
- data/lib/wurk/manager.rb +29 -13
- data/lib/wurk/metrics/history.rb +4 -3
- data/lib/wurk/metrics/queue_rollup.rb +13 -1
- data/lib/wurk/metrics/rollup.rb +13 -1
- data/lib/wurk/middleware/interrupt_handler.rb +7 -6
- data/lib/wurk/middleware/poison_pill.rb +70 -29
- data/lib/wurk/middleware.rb +2 -2
- data/lib/wurk/pool_checkout.rb +29 -0
- data/lib/wurk/process_set.rb +10 -5
- data/lib/wurk/processor.rb +6 -0
- data/lib/wurk/profiler.rb +3 -2
- data/lib/wurk/queue.rb +10 -7
- data/lib/wurk/rails_boot.rb +38 -7
- data/lib/wurk/redis_client_adapter.rb +48 -4
- data/lib/wurk/redis_options.rb +142 -0
- data/lib/wurk/redis_pool.rb +102 -39
- data/lib/wurk/scheduled.rb +30 -2
- data/lib/wurk/stats.rb +14 -9
- data/lib/wurk/swarm/child_boot.rb +12 -0
- data/lib/wurk/swarm.rb +174 -33
- data/lib/wurk/timer_loop.rb +14 -0
- data/lib/wurk/version.rb +1 -1
- data/lib/wurk/web/enterprise.rb +58 -6
- data/lib/wurk/web/extension.rb +1 -1
- data/lib/wurk/web/search.rb +5 -3
- data/lib/wurk.rb +53 -2
- data/vendor/assets/dashboard/assets/{BatchDetail-YRymNsrB.js → BatchDetail-OmC5NPgw.js} +1 -1
- data/vendor/assets/dashboard/assets/{Batches-HY4hHdQU.js → Batches-CIpai7St.js} +1 -1
- data/vendor/assets/dashboard/assets/{Busy-FCEN1Bpx.js → Busy-A_kwSR6Q.js} +1 -1
- data/vendor/assets/dashboard/assets/{Cron-DO3J2zcp.js → Cron-BG7HTqlp.js} +1 -1
- data/vendor/assets/dashboard/assets/{Dashboard-B9rOrkzk.js → Dashboard-A_ToqHoo.js} +1 -1
- data/vendor/assets/dashboard/assets/{Dead-Bi4GGk9a.js → Dead-8J21jMyK.js} +1 -1
- data/vendor/assets/dashboard/assets/Extension-B4Q9FIQu.js +1 -0
- data/vendor/assets/dashboard/assets/{FilterBox-IJkHYpdm.js → FilterBox-Fh_Ae7UW.js} +1 -1
- data/vendor/assets/dashboard/assets/{JobDetailModal-DS1ypyoc.js → JobDetailModal-Ceng0PMB.js} +1 -1
- data/vendor/assets/dashboard/assets/{Limiters-Nz7UbNeJ.js → Limiters-CruDWvNZ.js} +1 -1
- data/vendor/assets/dashboard/assets/{Metrics-BBTDxcaE.js → Metrics-CIT7VCoN.js} +1 -1
- data/vendor/assets/dashboard/assets/Modal-CN3rdKA_.js +1 -0
- data/vendor/assets/dashboard/assets/{Queues-D9PH_THs.js → Queues-D86FYohJ.js} +1 -1
- data/vendor/assets/dashboard/assets/{Retries-CAKzDgYG.js → Retries-Bz1O1D-i.js} +1 -1
- data/vendor/assets/dashboard/assets/{Scheduled-DTYw1X8S.js → Scheduled-B6h2akTu.js} +1 -1
- data/vendor/assets/dashboard/assets/{Search-c4vFxDG_.js → Search-OOu22e5s.js} +1 -1
- data/vendor/assets/dashboard/assets/index-BdiUEDXX.css +1 -0
- data/vendor/assets/dashboard/assets/index-D_lSDwKw.js +141 -0
- data/vendor/assets/dashboard/assets/{useResetPageOnEmpty-B_FsMah6.js → useResetPageOnEmpty-dVPGEWzn.js} +1 -1
- data/vendor/assets/dashboard/index.html +2 -2
- data/vendor/assets/dashboard/wurk-manifest.json +2 -2
- metadata +22 -20
- data/vendor/assets/dashboard/assets/Extension-BSv8ddW_.js +0 -1
- data/vendor/assets/dashboard/assets/Modal-Crrsu64-.js +0 -1
- data/vendor/assets/dashboard/assets/index-BxjdeuOa.css +0 -1
- data/vendor/assets/dashboard/assets/index-DQu7WY9y.js +0 -141
data/lib/wurk/heartbeat.rb
CHANGED
|
@@ -11,14 +11,17 @@ module Wurk
|
|
|
11
11
|
# Wurk::Launcher so the launcher can stay focused on lifecycle and so the
|
|
12
12
|
# heartbeat schema lives in one place readers can grep for.
|
|
13
13
|
#
|
|
14
|
-
# Each beat is
|
|
14
|
+
# Each beat is two pipelined round-trips. The identity write:
|
|
15
15
|
# SADD processes <identity>
|
|
16
16
|
# HSET <identity> info concurrency busy beat quiet rss rtt_us
|
|
17
17
|
# EXPIRE <identity> 60
|
|
18
18
|
# UNLINK <identity>:work
|
|
19
19
|
# HSET <identity>:work <tid> <json> ... (only if WORK_STATE non-empty)
|
|
20
20
|
# EXPIRE <identity>:work 60 (only if WORK_STATE non-empty)
|
|
21
|
+
# then the signal drain:
|
|
21
22
|
# LPOP <identity>-signals × BEAT_PAUSE
|
|
23
|
+
# Two rather than one because only the first is safe to replay — see
|
|
24
|
+
# #pipelined_beat and #drain_signals.
|
|
22
25
|
#
|
|
23
26
|
# The work hash is UNLINK-then-rewritten on every beat — a dropped beat
|
|
24
27
|
# momentarily empties it, and ProcessSet#cleanup compensates by SREM-ing
|
|
@@ -90,16 +93,17 @@ module Wurk
|
|
|
90
93
|
|
|
91
94
|
private
|
|
92
95
|
|
|
93
|
-
#
|
|
94
|
-
#
|
|
95
|
-
#
|
|
96
|
+
# Every command in #write_beat converges on the state this snapshot
|
|
97
|
+
# describes however many times it lands, so the beat claims apply-safety and
|
|
98
|
+
# rides out a blip the way it did before the pool started splitting replay
|
|
99
|
+
# by it. `rtt_us` measures this write alone — the signal drain that follows
|
|
100
|
+
# is a separate checkout and deliberately not part of the gauge.
|
|
96
101
|
def pipelined_beat
|
|
97
102
|
work_snapshot = Processor::WORK_STATE.dup
|
|
98
|
-
lead = 4 + (work_snapshot.empty? ? 0 : 2)
|
|
99
103
|
t0 = ::Process.clock_gettime(::Process::CLOCK_MONOTONIC, :microsecond)
|
|
100
|
-
|
|
104
|
+
redis(idempotent: true) { |conn| conn.pipelined { |pipe| write_beat(pipe, work_snapshot) } }
|
|
101
105
|
rtt = ::Process.clock_gettime(::Process::CLOCK_MONOTONIC, :microsecond) - t0
|
|
102
|
-
[
|
|
106
|
+
[drain_signals, rtt]
|
|
103
107
|
end
|
|
104
108
|
|
|
105
109
|
def write_beat(pipe, work_snapshot)
|
|
@@ -107,7 +111,6 @@ module Wurk
|
|
|
107
111
|
pipe.call('HSET', @identity, *beat_hash_args(work_snapshot.size))
|
|
108
112
|
pipe.call('EXPIRE', @identity, TTL_SECONDS)
|
|
109
113
|
write_work_hash(pipe, work_snapshot)
|
|
110
|
-
drain_signals(pipe)
|
|
111
114
|
end
|
|
112
115
|
|
|
113
116
|
def beat_hash_args(busy)
|
|
@@ -132,10 +135,19 @@ module Wurk
|
|
|
132
135
|
pipe.call('EXPIRE', work_key, TTL_SECONDS)
|
|
133
136
|
end
|
|
134
137
|
|
|
138
|
+
# Its own checkout, and never an idempotent one: LPOP is destructive and
|
|
139
|
+
# carries its result in the reply, so a replay after a lost reply discards
|
|
140
|
+
# whatever the first attempt already popped — and a discarded entry is a
|
|
141
|
+
# dashboard TERM or TSTP this process never acts on. Fused into the beat
|
|
142
|
+
# pipeline it would have dragged those writes down to the same no-replay
|
|
143
|
+
# default, or worse, invited a later sweep to claim the LPOPs alongside
|
|
144
|
+
# them. Runs after the beat so a signal only leaves Redis once the write
|
|
145
|
+
# that reports us alive has landed.
|
|
146
|
+
#
|
|
135
147
|
# LPOP one entry per second of cadence so a flood of queued signals
|
|
136
148
|
# can't stall the beat; anything older drains on the next beat.
|
|
137
|
-
def drain_signals
|
|
138
|
-
BEAT_PAUSE.times { pipe.call('LPOP', "#{@identity}-signals") }
|
|
149
|
+
def drain_signals
|
|
150
|
+
redis { |conn| conn.pipelined { |pipe| BEAT_PAUSE.times { pipe.call('LPOP', "#{@identity}-signals") } } }
|
|
139
151
|
end
|
|
140
152
|
|
|
141
153
|
def info_hash
|
data/lib/wurk/history.rb
CHANGED
|
@@ -71,11 +71,23 @@ module Wurk
|
|
|
71
71
|
end
|
|
72
72
|
|
|
73
73
|
def start
|
|
74
|
-
@thread
|
|
74
|
+
return @thread if @thread
|
|
75
|
+
|
|
76
|
+
@timer.reset
|
|
77
|
+
@thread = safe_thread('history-snapshot') { @timer.run { tick } }
|
|
75
78
|
end
|
|
76
79
|
|
|
80
|
+
# Blocks until the thread is really gone: the launcher releases the cluster
|
|
81
|
+
# lock immediately after this returns, and a snapshot still in flight would
|
|
82
|
+
# race the next leader's first one.
|
|
83
|
+
#
|
|
84
|
+
# Cleared only on a confirmed join (Thread#join returns nil on timeout): a
|
|
85
|
+
# wedged thread must stay tracked so #start's guard returns it instead of
|
|
86
|
+
# calling @timer.reset, which would un-terminate the loop it is still
|
|
87
|
+
# inside and leave two threads writing the same stream.
|
|
77
88
|
def terminate
|
|
78
89
|
@timer.terminate
|
|
90
|
+
@thread = nil if @thread&.join(TimerLoop::JOIN_TIMEOUT)
|
|
79
91
|
end
|
|
80
92
|
|
|
81
93
|
# Leader-gated: only the elected leader emits, so N workers don't each
|
data/lib/wurk/launcher.rb
CHANGED
|
@@ -13,6 +13,7 @@ require_relative 'metrics/rollup'
|
|
|
13
13
|
require_relative 'metrics/queue_rollup'
|
|
14
14
|
require_relative 'history'
|
|
15
15
|
require_relative 'fetcher/reaper'
|
|
16
|
+
require_relative 'timer_loop'
|
|
16
17
|
|
|
17
18
|
module Wurk
|
|
18
19
|
# Top-level supervisor inside each worker process. Owns the Manager pool
|
|
@@ -43,18 +44,23 @@ module Wurk
|
|
|
43
44
|
# (Sidekiq's drop-in surface). The single source of truth is Heartbeat.
|
|
44
45
|
BEAT_PAUSE = Heartbeat::BEAT_PAUSE
|
|
45
46
|
|
|
47
|
+
# Bound on how long #stop waits for the boot-time reclaim sweep before
|
|
48
|
+
# moving on — it can still be scanning a large keyspace when a fast
|
|
49
|
+
# shutdown lands right after boot; teardown must not hang on it.
|
|
50
|
+
BOOT_RECLAIM_JOIN_TIMEOUT = 5
|
|
51
|
+
|
|
46
52
|
attr_accessor :managers, :poller, :cron_poller, :metrics_rollup, :queue_rollup, :history
|
|
47
53
|
|
|
48
54
|
def initialize(config, embedded: false)
|
|
49
55
|
@config = config
|
|
50
56
|
@embedded = embedded
|
|
51
|
-
#
|
|
52
|
-
#
|
|
53
|
-
#
|
|
54
|
-
#
|
|
57
|
+
# @done is "quieted": stop fetching, stay alive, report quiet=true. It
|
|
58
|
+
# deliberately does NOT stop the heartbeat — a quieted process that stopped
|
|
59
|
+
# beating would never publish quiet=true and would expire out of the live
|
|
60
|
+
# set (#236). Only #stop ends the beat, by terminating @beat_timer.
|
|
55
61
|
@done = false
|
|
56
|
-
@
|
|
57
|
-
@managers =
|
|
62
|
+
@beat_timer = TimerLoop.new(BEAT_PAUSE)
|
|
63
|
+
@managers = build_managers
|
|
58
64
|
@poller = build_poller
|
|
59
65
|
@cron_poller = build_cron_poller
|
|
60
66
|
@metrics_rollup = build_metrics_rollup
|
|
@@ -87,18 +93,21 @@ module Wurk
|
|
|
87
93
|
@config.capsules.each_value(&:prepare!)
|
|
88
94
|
@config.freeze!
|
|
89
95
|
@heartbeat_thread = safe_thread('heartbeat', &method(:start_heartbeat)) if async_beat
|
|
90
|
-
@poller
|
|
91
|
-
@leader&.start
|
|
92
|
-
@cron_poller&.start
|
|
93
|
-
@metrics_rollup&.start
|
|
94
|
-
@queue_rollup&.start
|
|
95
|
-
@history&.start
|
|
96
|
+
[@poller, @leader, @cron_poller, @metrics_rollup, @queue_rollup, @history].compact.each(&:start)
|
|
96
97
|
@managers.each(&:start)
|
|
97
98
|
@reaper.start
|
|
98
99
|
# Run on a background thread so /ready probe isn't delayed by a large
|
|
99
100
|
# orphan sweep (reaper.reclaim! is atomic, but can scan many entries).
|
|
100
101
|
@boot_reclaim_thread = safe_thread('boot-reclaim', &method(:boot_reclaim))
|
|
101
102
|
@health_server&.start
|
|
103
|
+
rescue StandardError
|
|
104
|
+
# Boot is not atomic: whatever raised (a health-check port already bound,
|
|
105
|
+
# ThreadError at the OS thread limit) leaves the steps before it holding
|
|
106
|
+
# threads, sockets and a leader campaign — and the caller is about to drop
|
|
107
|
+
# its only reference to us, so nothing else can ever release them. Guarded,
|
|
108
|
+
# because the caller must see the boot failure, not a rollback failure.
|
|
109
|
+
teardown_step('boot-rollback') { stop }
|
|
110
|
+
raise
|
|
102
111
|
end
|
|
103
112
|
|
|
104
113
|
# Idempotent. Flips `stopping?` true, halts fetching across every
|
|
@@ -118,26 +127,18 @@ module Wurk
|
|
|
118
127
|
|
|
119
128
|
# Graceful shutdown. Deadline is monotonic so wall-clock skew can't
|
|
120
129
|
# extend it. Managers stop in parallel threads so a slow capsule
|
|
121
|
-
# doesn't block its siblings
|
|
130
|
+
# doesn't block its siblings — and each drain is guarded, because a
|
|
131
|
+
# capsule that blows up mid-drain (Redis down during bulk_requeue) used
|
|
132
|
+
# to surface out of `join` and skip the whole teardown tail, leaving the
|
|
133
|
+
# leader lock held, the process listed as live, and its port open.
|
|
122
134
|
def stop
|
|
123
135
|
deadline = ::Process.clock_gettime(::Process::CLOCK_MONOTONIC) + (@config[:timeout] || 25)
|
|
124
136
|
quiet
|
|
125
|
-
stoppers = @managers.map { |m| Thread.new { m.stop(deadline) } }
|
|
137
|
+
stoppers = @managers.map { |m| Thread.new { teardown_step('manager') { m.stop(deadline) } } }
|
|
126
138
|
fire_event(:shutdown, reverse: true)
|
|
127
139
|
stoppers.each(&:join)
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
@cron_poller&.terminate
|
|
131
|
-
@metrics_rollup&.terminate
|
|
132
|
-
@queue_rollup&.terminate
|
|
133
|
-
@history&.terminate
|
|
134
|
-
@reaper&.stop
|
|
135
|
-
# CAS-release the cluster lock now (planned shutdown) so a follower can
|
|
136
|
-
# take over immediately instead of waiting out the TTL.
|
|
137
|
-
@leader&.stop
|
|
138
|
-
stop_heartbeat
|
|
139
|
-
clear_heartbeat
|
|
140
|
-
fire_event(:exit, reverse: true)
|
|
140
|
+
ensure
|
|
141
|
+
release_components
|
|
141
142
|
end
|
|
142
143
|
|
|
143
144
|
def stopping?
|
|
@@ -162,9 +163,12 @@ module Wurk
|
|
|
162
163
|
|
|
163
164
|
write_stats(processed, failed, expired)
|
|
164
165
|
rescue StandardError => e
|
|
165
|
-
#
|
|
166
|
-
#
|
|
167
|
-
#
|
|
166
|
+
# The counters were reset above, so a raise here drops this window's stats
|
|
167
|
+
# for good: #write_stats cannot claim apply-safety and the pool no longer
|
|
168
|
+
# replays it through a blip. Accepted deliberately — adding them back
|
|
169
|
+
# would double-count the case where the INCRBYs did land and only the
|
|
170
|
+
# reply was lost, and the per-job at-least-once semantics don't apply to
|
|
171
|
+
# *counters*. The next beat resets again.
|
|
168
172
|
handle_exception(e, { context: 'flush_stats' })
|
|
169
173
|
end
|
|
170
174
|
|
|
@@ -174,6 +178,60 @@ module Wurk
|
|
|
174
178
|
|
|
175
179
|
private
|
|
176
180
|
|
|
181
|
+
# Teardown tail, driven from #stop's ensure. Every release is guarded on
|
|
182
|
+
# its own because they are independent: a Redis blip in one (leader CAS
|
|
183
|
+
# release, heartbeat clear) must not strand the ones after it — the process
|
|
184
|
+
# is exiting either way, so the only thing worse than a failed release is a
|
|
185
|
+
# skipped one. Order matters twice over: full shutdown stops the periodic
|
|
186
|
+
# loops (they survived #quiet) before the cluster lock is CAS-released, so
|
|
187
|
+
# no tick races a follower's promotion, and the release itself happens on a
|
|
188
|
+
# planned shutdown rather than waiting out the lock TTL.
|
|
189
|
+
def release_components
|
|
190
|
+
# Joined first, before anything below tears down the Redis pool it's
|
|
191
|
+
# still reading from: it started as a fire-and-forget thread in #run
|
|
192
|
+
# and a shutdown landing right after boot could otherwise race a
|
|
193
|
+
# pool disconnect mid-scan. Bounded: a full-keyspace scan can outlast
|
|
194
|
+
# the timeout, in which case the thread is left running rather than
|
|
195
|
+
# blocking shutdown on it — #boot_reclaim already rescues and logs on
|
|
196
|
+
# its own, so a straggler racing #reset_redis_pools! below just means
|
|
197
|
+
# a noisy log line, not a hang.
|
|
198
|
+
teardown_step('boot-reclaim-join') { @boot_reclaim_thread&.join(BOOT_RECLAIM_JOIN_TIMEOUT) }
|
|
199
|
+
stop_periodic_components
|
|
200
|
+
%i[stop_heartbeat clear_heartbeat].each { |step| teardown_step(step) { send(step) } }
|
|
201
|
+
# Unguarded: fire_event already reports and skips past a raising hook.
|
|
202
|
+
fire_event(:exit, reverse: true)
|
|
203
|
+
# Embedded only: a swarm child or standalone process exits right after
|
|
204
|
+
# #stop anyway, so disconnecting here would just make every unit test
|
|
205
|
+
# that inspects Redis post-stop rebuild a pool for no reason. Embedded
|
|
206
|
+
# hosts (Puma, a rake task) keep running and can `run` again later —
|
|
207
|
+
# without this a stop-then-run cycle doubles the live socket set.
|
|
208
|
+
teardown_step('redis-pools') { @config.reset_redis_pools! } if @embedded
|
|
209
|
+
ensure
|
|
210
|
+
# In an ensure of its own, not merely a guarded step: a leaked TCPServer
|
|
211
|
+
# FD survives even a non-StandardError unwind (a second TERM landing
|
|
212
|
+
# mid-teardown), and kubelet would keep getting 200s from a process that
|
|
213
|
+
# is already gone.
|
|
214
|
+
teardown_step('health-server') { @health_server&.stop }
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
# Split out of #release_components to keep it under the AbcSize/
|
|
218
|
+
# CyclomaticComplexity ceilings — these three are the "periodic loop"
|
|
219
|
+
# releases, independently guarded like everything else in the tail.
|
|
220
|
+
def stop_periodic_components
|
|
221
|
+
[@cron_poller, @metrics_rollup, @queue_rollup, @history].each { |t| teardown_step(t.class) { t&.terminate } }
|
|
222
|
+
teardown_step('reaper') { @reaper&.stop }
|
|
223
|
+
teardown_step('leader') { @leader&.stop }
|
|
224
|
+
end
|
|
225
|
+
|
|
226
|
+
def teardown_step(label)
|
|
227
|
+
yield
|
|
228
|
+
rescue StandardError => e
|
|
229
|
+
handle_exception(e, { context: "launcher-stop-#{label}" })
|
|
230
|
+
end
|
|
231
|
+
|
|
232
|
+
# INCRBY is additive, so a pipeline replayed after a lost reply counts this
|
|
233
|
+
# window twice on every `stat:` key it touches — no apply-safety claim. (The
|
|
234
|
+
# EXPIREs riding along are harmless to repeat; the INCRBYs pin the block.)
|
|
177
235
|
def write_stats(processed, failed, expired)
|
|
178
236
|
day = Time.now.utc.strftime('%F')
|
|
179
237
|
@config.redis do |conn|
|
|
@@ -217,74 +275,83 @@ module Wurk
|
|
|
217
275
|
|
|
218
276
|
# Erase the live-process footprint. flush_stats first so we don't drop
|
|
219
277
|
# the final batch of counters; then Heartbeat#stop! removes us from the
|
|
220
|
-
# `processes` SET and UNLINK-s the identity + work hashes.
|
|
221
|
-
# server is closed alongside so kubelet stops getting 200s after the
|
|
222
|
-
# process is no longer healthy.
|
|
278
|
+
# `processes` SET and UNLINK-s the identity + work hashes.
|
|
223
279
|
def clear_heartbeat
|
|
224
280
|
flush_stats
|
|
225
281
|
@heartbeat&.stop!
|
|
226
|
-
@health_server&.stop
|
|
227
282
|
end
|
|
228
283
|
|
|
229
284
|
# Terminate the heartbeat loop and wait for it to exit before clear_heartbeat
|
|
230
285
|
# removes us from the `processes` SET — otherwise a final in-flight beat could
|
|
231
|
-
# SADD us back right after the SREM
|
|
232
|
-
#
|
|
286
|
+
# SADD us back right after the SREM and the identity would linger for a full
|
|
287
|
+
# 60s TTL.
|
|
288
|
+
#
|
|
289
|
+
# The join is unbounded on purpose. This used to be `wakeup` + `join(BEAT_PAUSE)`,
|
|
290
|
+
# which lost the race whenever the beat was mid-Redis-call: `wakeup` does nothing
|
|
291
|
+
# to a thread that isn't sleeping, so the loop then slept a full interval and the
|
|
292
|
+
# bounded join returned with the thread still live — exactly the resurrect-after-
|
|
293
|
+
# SREM this ordering exists to prevent. A condvar can't be missed (terminate flips
|
|
294
|
+
# the flag inside the critical section the loop re-checks it in), so all that is
|
|
295
|
+
# left to wait out is one in-flight beat, whose Redis calls are timeout-bounded.
|
|
233
296
|
def stop_heartbeat
|
|
234
|
-
@
|
|
297
|
+
@beat_timer.terminate
|
|
235
298
|
thread = @heartbeat_thread
|
|
236
299
|
return unless thread
|
|
237
|
-
# Embedded dashboard-TERM
|
|
238
|
-
# raises ThreadError.
|
|
300
|
+
# Embedded dashboard-TERM can run `stop` from the beat itself; a self-join
|
|
301
|
+
# raises ThreadError. The timer is terminated, so the loop exits after this beat.
|
|
239
302
|
return if thread == Thread.current
|
|
240
303
|
|
|
241
|
-
|
|
242
|
-
thread.wakeup
|
|
243
|
-
rescue ThreadError
|
|
244
|
-
nil
|
|
245
|
-
end
|
|
246
|
-
thread.join(BEAT_PAUSE)
|
|
304
|
+
thread.join
|
|
247
305
|
end
|
|
248
306
|
|
|
249
|
-
# Heartbeat thread loop. `safe_thread` already wraps exceptions.
|
|
250
|
-
#
|
|
251
|
-
#
|
|
252
|
-
#
|
|
253
|
-
#
|
|
307
|
+
# Heartbeat thread loop. `safe_thread` already wraps exceptions. Beats once up
|
|
308
|
+
# front — TimerLoop#run waits before its first yield, and the dashboard has to
|
|
309
|
+
# see this process the moment it can pick up jobs — then ticks until #stop
|
|
310
|
+
# terminates the timer. Note it does NOT stop on `@done`: a *quieted* process
|
|
311
|
+
# keeps beating so it publishes `quiet=true` instead of vanishing (#236).
|
|
254
312
|
def start_heartbeat
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
sleep BEAT_PAUSE
|
|
258
|
-
end
|
|
313
|
+
heartbeat
|
|
314
|
+
@beat_timer.run { heartbeat }
|
|
259
315
|
logger.info('Heartbeat stopping...')
|
|
260
316
|
end
|
|
261
317
|
|
|
262
318
|
# Dashboard-queued signals must behave exactly like OS signals, so a
|
|
263
|
-
# standalone process re-delivers to itself and lets the installed trap
|
|
264
|
-
#
|
|
265
|
-
#
|
|
266
|
-
#
|
|
267
|
-
#
|
|
268
|
-
# its own thread because `stop` joins the heartbeat thread we're on.
|
|
319
|
+
# standalone process re-delivers to itself and lets the installed trap run —
|
|
320
|
+
# that wakes the main thread (CLI self-pipe / child dispatcher) so the
|
|
321
|
+
# process actually exits instead of stopping its managers and then parking
|
|
322
|
+
# forever. Embedded mode owns no traps (and self-TERM would kill the host
|
|
323
|
+
# app), so quiet applies directly.
|
|
269
324
|
def dispatch_signal(sig)
|
|
270
325
|
case sig
|
|
271
|
-
when 'TSTP'
|
|
272
|
-
|
|
273
|
-
sig == 'TSTP' ? quiet : Thread.new { stop }
|
|
274
|
-
else
|
|
275
|
-
redeliver(sig)
|
|
276
|
-
end
|
|
326
|
+
when 'TSTP' then @embedded ? quiet : redeliver(sig)
|
|
327
|
+
when 'TERM' then request_shutdown
|
|
277
328
|
else
|
|
278
329
|
logger.warn { "Unknown signal in #{identity}-signals: #{sig.inspect}" }
|
|
279
330
|
end
|
|
280
331
|
end
|
|
281
332
|
|
|
333
|
+
# The one way anything inside this process asks it to shut down gracefully:
|
|
334
|
+
# dashboard-queued TERM, or a Manager that can no longer hold its
|
|
335
|
+
# concurrency. Standalone hands off to the installed TERM trap; embedded
|
|
336
|
+
# owns no traps, so it drains in place — on its own thread, because callers
|
|
337
|
+
# may be a thread `stop` itself joins (the heartbeat) or kills (a Processor).
|
|
338
|
+
def request_shutdown
|
|
339
|
+
@embedded ? Thread.new { stop } : redeliver('TERM')
|
|
340
|
+
end
|
|
341
|
+
|
|
282
342
|
# Separate method so tests can stub it — really sending TERM/TSTP would
|
|
283
343
|
# kill or suspend the test process.
|
|
284
344
|
def redeliver(sig)
|
|
285
345
|
::Process.kill(sig, ::Process.pid)
|
|
286
346
|
end
|
|
287
347
|
|
|
348
|
+
# One Manager per capsule, each holding our shutdown request: a Manager
|
|
349
|
+
# that can no longer replace a dead Processor has to take the process
|
|
350
|
+
# down, and only the Launcher knows how this process exits.
|
|
351
|
+
def build_managers
|
|
352
|
+
@config.capsules.values.map { |cap| Manager.new(cap, shutdown: method(:request_shutdown)) }
|
|
353
|
+
end
|
|
354
|
+
|
|
288
355
|
def build_poller
|
|
289
356
|
Wurk::Scheduled::Poller.new(@config)
|
|
290
357
|
end
|
data/lib/wurk/leader.rb
CHANGED
|
@@ -3,6 +3,8 @@
|
|
|
3
3
|
require 'securerandom'
|
|
4
4
|
require 'socket'
|
|
5
5
|
require_relative 'component'
|
|
6
|
+
require_relative 'lua'
|
|
7
|
+
require_relative 'pool_checkout'
|
|
6
8
|
|
|
7
9
|
module Wurk
|
|
8
10
|
# Cluster leader election via Redis `SET NX EX`. Single-leader-per-cluster
|
|
@@ -89,9 +91,19 @@ module Wurk
|
|
|
89
91
|
|
|
90
92
|
# CAS DEL — only drop the key if we still own it, otherwise a stale
|
|
91
93
|
# release would yank leadership from whichever follower took over.
|
|
94
|
+
# GET-then-DEL is not that CAS: our key can lapse between the two
|
|
95
|
+
# commands and a follower can win the election in the gap, whereupon
|
|
96
|
+
# the bare DEL deletes *its* lock and leaves the cluster leaderless
|
|
97
|
+
# (or, once the ex-leader re-campaigns, doubly led) for up to one
|
|
98
|
+
# renew interval — double cron fires, double rollups. The compare and
|
|
99
|
+
# the delete happen in one server-side step instead, via the script
|
|
100
|
+
# `Unique.release_if_owner` already shares.
|
|
101
|
+
#
|
|
102
|
+
# `idempotent:` — a replay after a lost reply finds the key either
|
|
103
|
+
# already gone or no longer ours, so it can only be a no-op.
|
|
92
104
|
def release
|
|
93
|
-
redis_call do |c|
|
|
94
|
-
|
|
105
|
+
redis_call(idempotent: true) do |c|
|
|
106
|
+
Wurk::Lua::Loader.eval_cached(c, :release_if_owner, keys: [@key], argv: [@owner])
|
|
95
107
|
end
|
|
96
108
|
@held = false
|
|
97
109
|
@token = nil
|
|
@@ -107,6 +119,11 @@ module Wurk
|
|
|
107
119
|
# follower, it polls every `follower_interval`. Caller must invoke
|
|
108
120
|
# `stop` for orderly shutdown — the thread also releases its lock on
|
|
109
121
|
# exit.
|
|
122
|
+
#
|
|
123
|
+
# The spawn happens *inside* the lock: with it outside, two callers
|
|
124
|
+
# racing into `start` both clear the nil check and both spawn, and the
|
|
125
|
+
# loser's loop is never recorded — an unjoinable second campaigner that
|
|
126
|
+
# outlives `stop` and keeps renewing the cluster lock.
|
|
110
127
|
def start
|
|
111
128
|
return nil if disabled?
|
|
112
129
|
|
|
@@ -114,17 +131,19 @@ module Wurk
|
|
|
114
131
|
return @thread if @thread
|
|
115
132
|
|
|
116
133
|
@done = false
|
|
134
|
+
@thread = spawn_loop_thread
|
|
117
135
|
end
|
|
118
|
-
@thread
|
|
136
|
+
@thread
|
|
119
137
|
end
|
|
120
138
|
|
|
121
139
|
def stop
|
|
122
|
-
@mutex.synchronize do
|
|
140
|
+
thread = @mutex.synchronize do
|
|
123
141
|
@done = true
|
|
124
142
|
@sleeper.signal
|
|
143
|
+
@thread
|
|
125
144
|
end
|
|
126
|
-
|
|
127
|
-
@thread = nil
|
|
145
|
+
thread&.join
|
|
146
|
+
@mutex.synchronize { @thread = nil }
|
|
128
147
|
release
|
|
129
148
|
end
|
|
130
149
|
|
|
@@ -175,13 +194,13 @@ module Wurk
|
|
|
175
194
|
@config.handle_exception(e, event: :leader) if @config.respond_to?(:handle_exception)
|
|
176
195
|
end
|
|
177
196
|
|
|
178
|
-
def redis_call(&)
|
|
197
|
+
def redis_call(idempotent: false, &)
|
|
179
198
|
if @pool
|
|
180
|
-
|
|
199
|
+
PoolCheckout.with(@pool, idempotent, &)
|
|
181
200
|
elsif @config
|
|
182
|
-
@config.redis(&)
|
|
201
|
+
@config.redis(idempotent:, &)
|
|
183
202
|
else
|
|
184
|
-
Wurk.redis(&)
|
|
203
|
+
Wurk.redis(idempotent:, &)
|
|
185
204
|
end
|
|
186
205
|
end
|
|
187
206
|
|
data/lib/wurk/limiter/base.rb
CHANGED
|
@@ -104,17 +104,15 @@ module Wurk
|
|
|
104
104
|
end
|
|
105
105
|
end
|
|
106
106
|
|
|
107
|
+
# Metadata HSET + EXPIRE + `lmtr-list` SADD in a single round trip. The
|
|
108
|
+
# script also gates the SADD on the metadata being newly written, so a
|
|
109
|
+
# per-job construction (`stripe-#{user_id}`) stops rewriting a set that
|
|
110
|
+
# grows one member per name — see limiter_register.lua for why that gate
|
|
111
|
+
# is only safe while it is atomic with the HSET.
|
|
107
112
|
def register!
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
'HSET', meta_key,
|
|
112
|
-
'type', type.to_s,
|
|
113
|
-
'options', JSON.dump(serializable_options),
|
|
114
|
-
'fingerprint', fingerprint
|
|
115
|
-
)
|
|
116
|
-
c.call('EXPIRE', meta_key, @options[:ttl])
|
|
117
|
-
end
|
|
113
|
+
lua(:limiter_register,
|
|
114
|
+
keys: [meta_key, LIST_KEY],
|
|
115
|
+
argv: [@name, type.to_s, JSON.dump(serializable_options), fingerprint, ttl])
|
|
118
116
|
end
|
|
119
117
|
|
|
120
118
|
def ttl
|
data/lib/wurk/limiter/bucket.rb
CHANGED
|
@@ -6,7 +6,7 @@ module Wurk
|
|
|
6
6
|
module Limiter
|
|
7
7
|
# Atomic slot acquisition in a ZSET. Score = expiry epoch; the acquire
|
|
8
8
|
# script first evicts expired slots (bumping the `reclaimed` metric)
|
|
9
|
-
# then ZADDs if there's headroom.
|
|
9
|
+
# then ZADDs if there's headroom (bumping the `held` metric).
|
|
10
10
|
#
|
|
11
11
|
# On exhaustion: spin loop with backoff. The spec says "blocks via
|
|
12
12
|
# Redis stream XREAD" — that's a perf optimization; the visible
|
|
@@ -38,33 +38,15 @@ module Wurk
|
|
|
38
38
|
raise ArgumentError, 'block required' unless block
|
|
39
39
|
|
|
40
40
|
started = monotime
|
|
41
|
-
deadline = started + @options[:wait_timeout]
|
|
42
41
|
slot = random_id
|
|
43
|
-
acquired_at =
|
|
44
|
-
|
|
45
|
-
result = acquire(slot)
|
|
46
|
-
if result[0].to_i == 1
|
|
47
|
-
acquired_at = monotime
|
|
48
|
-
break
|
|
49
|
-
end
|
|
50
|
-
|
|
51
|
-
return if @options[:policy] == :ignore
|
|
52
|
-
|
|
53
|
-
remaining = deadline - monotime
|
|
54
|
-
if remaining <= 0
|
|
55
|
-
bump_counter('overages')
|
|
56
|
-
raise OverLimit, self
|
|
57
|
-
end
|
|
58
|
-
|
|
59
|
-
sleep [remaining, WAIT_SLEEP].min
|
|
60
|
-
end
|
|
42
|
+
acquired_at = wait_for_slot(slot, started + @options[:wait_timeout])
|
|
43
|
+
return unless acquired_at
|
|
61
44
|
|
|
62
45
|
begin
|
|
63
46
|
incr_immediate_or_waited(acquired_at - started)
|
|
64
47
|
block.call
|
|
65
48
|
ensure
|
|
66
|
-
|
|
67
|
-
bump_counter('held_time', (monotime - acquired_at).to_i) if acquired_at
|
|
49
|
+
record_release(slot, acquired_at)
|
|
68
50
|
end
|
|
69
51
|
end
|
|
70
52
|
|
|
@@ -98,6 +80,29 @@ module Wurk
|
|
|
98
80
|
"lmtr-stats:#{@name}"
|
|
99
81
|
end
|
|
100
82
|
|
|
83
|
+
# A ZREM that removes nothing means an acquirer already evicted our slot:
|
|
84
|
+
# we outran `lock_timeout`, which is what `overages` counts. Exhausting
|
|
85
|
+
# `wait_timeout` is a different event — it raises OverLimit and the server
|
|
86
|
+
# middleware counts it in the job's `overrated` field.
|
|
87
|
+
def record_release(slot, acquired_at)
|
|
88
|
+
bump_counter('overages') if release(slot).to_i.zero?
|
|
89
|
+
bump_counter('held_time', (monotime - acquired_at).to_i)
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
# Spins until a slot frees up. Returns the monotonic acquire time, nil
|
|
93
|
+
# when `policy: :ignore` gives up; raises OverLimit past `wait_timeout`.
|
|
94
|
+
def wait_for_slot(slot, deadline)
|
|
95
|
+
loop do
|
|
96
|
+
return monotime if acquire(slot)[0].to_i == 1
|
|
97
|
+
return if @options[:policy] == :ignore
|
|
98
|
+
|
|
99
|
+
remaining = deadline - monotime
|
|
100
|
+
raise OverLimit, self if remaining <= 0
|
|
101
|
+
|
|
102
|
+
sleep [remaining, WAIT_SLEEP].min
|
|
103
|
+
end
|
|
104
|
+
end
|
|
105
|
+
|
|
101
106
|
def acquire(slot)
|
|
102
107
|
lua(:limiter_concurrent_acquire,
|
|
103
108
|
keys: [state_key, stats_key],
|
data/lib/wurk/limiter/window.rb
CHANGED
|
@@ -19,18 +19,15 @@ module Wurk
|
|
|
19
19
|
end
|
|
20
20
|
|
|
21
21
|
def size
|
|
22
|
-
|
|
23
|
-
Wurk::Limiter.redis do |c|
|
|
24
|
-
c.call('ZREMRANGEBYSCORE', state_key, '-inf', "(#{cutoff}")
|
|
25
|
-
c.call('ZCARD', state_key).to_i
|
|
26
|
-
end
|
|
22
|
+
window_state.first
|
|
27
23
|
end
|
|
28
24
|
|
|
29
25
|
# used = entries still inside the window; limit = count; reset_at =
|
|
30
26
|
# when the oldest entry slides out (freeing a slot), or nil when
|
|
31
27
|
# idle (#16).
|
|
32
28
|
def status
|
|
33
|
-
|
|
29
|
+
used, oldest = window_state
|
|
30
|
+
build_status(used: used, limit: @options[:count], reset_at: oldest && (oldest + interval_seconds))
|
|
34
31
|
end
|
|
35
32
|
|
|
36
33
|
def within_limit(used: 1, &block)
|
|
@@ -60,11 +57,16 @@ module Wurk
|
|
|
60
57
|
"lmtr-w:#{@name}"
|
|
61
58
|
end
|
|
62
59
|
|
|
63
|
-
#
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
60
|
+
# Count + oldest in-window score, both scoped by the Redis clock, in one
|
|
61
|
+
# read-only round trip. Reads never trim (the acquire script owns that):
|
|
62
|
+
# a dashboard host whose clock ran ahead used to evict live entries just
|
|
63
|
+
# by rendering `status`, freeing the window for a second full charge.
|
|
64
|
+
# Oldest timestamp + interval = the moment it leaves the window; -1 =
|
|
65
|
+
# nothing in window.
|
|
66
|
+
def window_state
|
|
67
|
+
count, oldest = lua(:limiter_window_status, keys: [state_key], argv: [interval_seconds])
|
|
68
|
+
score = oldest.to_f
|
|
69
|
+
[count.to_i, score.negative? ? nil : score]
|
|
68
70
|
end
|
|
69
71
|
|
|
70
72
|
def interval_seconds
|