wurk 1.1.0 → 1.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +25 -0
- data/app/controllers/concerns/wurk/same_origin_guard.rb +40 -0
- data/app/controllers/concerns/wurk/sse_streaming.rb +48 -0
- data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +53 -0
- data/app/controllers/wurk/api/pagination.rb +60 -11
- data/app/controllers/wurk/api/serializers.rb +5 -1
- data/app/controllers/wurk/api_controller.rb +51 -70
- data/app/controllers/wurk/application_controller.rb +26 -0
- data/app/controllers/wurk/dashboard_controller.rb +23 -1
- data/app/controllers/wurk/extensions_controller.rb +3 -12
- data/app/controllers/wurk/profiles_controller.rb +6 -1
- data/lib/wurk/batch/death_handler.rb +13 -0
- data/lib/wurk/batch/server_middleware.rb +9 -5
- data/lib/wurk/batch.rb +3 -0
- data/lib/wurk/capsule.rb +34 -21
- data/lib/wurk/cli.rb +16 -3
- data/lib/wurk/client/buffered.rb +30 -10
- data/lib/wurk/client.rb +45 -1
- data/lib/wurk/component.rb +23 -9
- data/lib/wurk/configuration.rb +83 -18
- data/lib/wurk/cron.rb +13 -1
- data/lib/wurk/dead_set.rb +16 -1
- data/lib/wurk/engine.rb +17 -1
- data/lib/wurk/fetcher/reliable.rb +46 -13
- data/lib/wurk/fetcher.rb +5 -0
- data/lib/wurk/health.rb +74 -22
- data/lib/wurk/history.rb +4 -21
- data/lib/wurk/job_retry.rb +3 -5
- data/lib/wurk/launcher.rb +26 -3
- data/lib/wurk/limiter/server_middleware.rb +16 -1
- data/lib/wurk/limiter.rb +6 -5
- data/lib/wurk/lua/loader.rb +11 -5
- data/lib/wurk/lua.rb +100 -9
- data/lib/wurk/manager.rb +59 -19
- data/lib/wurk/metrics/history.rb +24 -38
- data/lib/wurk/metrics/queue_rollup.rb +4 -21
- data/lib/wurk/metrics/rollup.rb +4 -21
- data/lib/wurk/processor.rb +8 -8
- data/lib/wurk/profile_set.rb +21 -6
- data/lib/wurk/profiler.rb +7 -5
- data/lib/wurk/rails_boot.rb +176 -0
- data/lib/wurk/railtie.rb +19 -47
- data/lib/wurk/redis_connection.rb +6 -9
- data/lib/wurk/redis_pool.rb +148 -28
- data/lib/wurk/scheduled.rb +35 -12
- data/lib/wurk/swarm/backoff.rb +70 -0
- data/lib/wurk/swarm/child_boot.rb +92 -13
- data/lib/wurk/swarm/orphan_guard.rb +105 -0
- data/lib/wurk/swarm/restart.rb +196 -0
- data/lib/wurk/swarm.rb +194 -78
- data/lib/wurk/timer_loop.rb +49 -0
- data/lib/wurk/version.rb +1 -1
- data/lib/wurk/web/extension.rb +4 -1
- data/lib/wurk/web/pool_scope.rb +46 -0
- data/lib/wurk/web/rack_app.rb +2 -1
- data/lib/wurk/web/search.rb +77 -18
- data/lib/wurk/web.rb +1 -0
- data/lib/wurk/worker/setter.rb +6 -1
- data/lib/wurk.rb +6 -2
- data/vendor/assets/dashboard/assets/ArgsValue-DYfBiXrJ.js +1 -0
- data/vendor/assets/dashboard/assets/BatchDetail-DTZ2HzcD.js +1 -0
- data/vendor/assets/dashboard/assets/Batches-BrnXA332.js +1 -0
- data/vendor/assets/dashboard/assets/Busy-C57G8Xb3.js +1 -0
- data/vendor/assets/dashboard/assets/Cron-DlyH88oo.js +1 -0
- data/vendor/assets/dashboard/assets/Dashboard-B887pxWf.js +1 -0
- data/vendor/assets/dashboard/assets/Dead-BPA7gs-X.js +1 -0
- data/vendor/assets/dashboard/assets/Extension-Bunf6XuU.js +1 -0
- data/vendor/assets/dashboard/assets/FilterBox-FCDi4ZCU.js +1 -0
- data/vendor/assets/dashboard/assets/JobDetailModal-DuMdKUMm.js +2 -0
- data/vendor/assets/dashboard/assets/Limiters-Br0aCPMK.js +1 -0
- data/vendor/assets/dashboard/assets/Metrics-DxBmuywH.js +1 -0
- data/vendor/assets/dashboard/assets/Modal-t4FI_LaY.js +1 -0
- data/vendor/assets/dashboard/assets/PageHeader-CsDvJSOA.js +1 -0
- data/vendor/assets/dashboard/assets/Profiles-Bkhoqjlq.js +1 -0
- data/vendor/assets/dashboard/assets/Queues-BvhA-vfI.js +1 -0
- data/vendor/assets/dashboard/assets/Retries-JEpB-1Yl.js +1 -0
- data/vendor/assets/dashboard/assets/Scheduled-DN_FbSwP.js +1 -0
- data/vendor/assets/dashboard/assets/Search-DJuK0YCJ.js +1 -0
- data/vendor/assets/dashboard/assets/Skeleton-DOYDkzg1.js +1 -0
- data/vendor/assets/dashboard/assets/charts-CVK0zAnC.js +1 -0
- data/vendor/assets/dashboard/assets/index-BnPX9Ptn.css +1 -0
- data/vendor/assets/dashboard/assets/index-CZTcs-pM.js +141 -0
- data/vendor/assets/dashboard/assets/useResetPageOnEmpty-CoZU4b3a.js +1 -0
- data/vendor/assets/dashboard/assets/useSort-wQcnbdsa.js +1 -0
- data/vendor/assets/dashboard/index.html +3 -3
- data/vendor/assets/dashboard/wurk-manifest.json +2 -2
- metadata +55 -26
- data/vendor/assets/dashboard/assets/ArgsValue-BUqJa-eG.js +0 -1
- data/vendor/assets/dashboard/assets/BatchDetail-C5dkqAzp.js +0 -1
- data/vendor/assets/dashboard/assets/Batches-bgkXn8tc.js +0 -1
- data/vendor/assets/dashboard/assets/Busy-QSHBFXhy.js +0 -1
- data/vendor/assets/dashboard/assets/Cron-CUHppvTA.js +0 -1
- data/vendor/assets/dashboard/assets/Dashboard-CzPudckV.js +0 -1
- data/vendor/assets/dashboard/assets/Dead-BIq4Nz_i.js +0 -1
- data/vendor/assets/dashboard/assets/Extension-CW36i9R1.js +0 -1
- data/vendor/assets/dashboard/assets/JobDetailModal-BhNdWSp7.js +0 -2
- data/vendor/assets/dashboard/assets/Limiters-CiI_DfUK.js +0 -1
- data/vendor/assets/dashboard/assets/Metrics-BzZ8ugms.js +0 -1
- data/vendor/assets/dashboard/assets/Modal-DzVfgsSF.js +0 -1
- data/vendor/assets/dashboard/assets/PageHeader-Dp3qhX3e.js +0 -1
- data/vendor/assets/dashboard/assets/Profiles-BTtIlTdR.js +0 -1
- data/vendor/assets/dashboard/assets/Queues-BuXoxQ4W.js +0 -1
- data/vendor/assets/dashboard/assets/Retries-D4HAPaOQ.js +0 -1
- data/vendor/assets/dashboard/assets/Scheduled-6NCZYVJh.js +0 -1
- data/vendor/assets/dashboard/assets/Search-JaB_-52c.js +0 -1
- data/vendor/assets/dashboard/assets/charts-6uvCyY0x.js +0 -1
- data/vendor/assets/dashboard/assets/i18n-gIeA5VLo.js +0 -1
- data/vendor/assets/dashboard/assets/index-BDG9tvBA.css +0 -1
- data/vendor/assets/dashboard/assets/index-DWfWAGBc.js +0 -141
- data/vendor/assets/dashboard/assets/useJobSetActions-DTDaAoWr.js +0 -1
- data/vendor/assets/dashboard/assets/usePageParam-BcAvRko-.js +0 -1
- data/vendor/assets/dashboard/assets/useSort-D5Am4bGq.js +0 -1
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
require 'socket'
|
|
4
4
|
require_relative '../component'
|
|
5
5
|
require_relative '../keys'
|
|
6
|
+
require_relative '../lua'
|
|
6
7
|
require_relative '../fetcher'
|
|
7
8
|
|
|
8
9
|
module Wurk
|
|
@@ -79,19 +80,20 @@ module Wurk
|
|
|
79
80
|
end
|
|
80
81
|
|
|
81
82
|
# Called on shutdown for jobs the Processor couldn't finish in time.
|
|
82
|
-
#
|
|
83
|
-
#
|
|
83
|
+
# Atomically moves each still-private UoW back to its public queue via
|
|
84
|
+
# the RELIABLE_REQUEUE Lua (LREM-guarded RPUSH): the job leaves the
|
|
85
|
+
# per-process private list and reappears on the public queue in one hop,
|
|
86
|
+
# so it's visible immediately after a deploy instead of waiting for the
|
|
87
|
+
# next boot's reaper. The guard makes the move idempotent against the
|
|
88
|
+
# cross-thread `job`-read race in Manager#hard_shutdown — a Processor
|
|
89
|
+
# that ACKed in that window is a no-op (LREM misses, RPUSH skipped), so a
|
|
90
|
+
# finished job is never resurrected. Sidekiq Pro super_fetch §3 retains
|
|
91
|
+
# in-flight in the private list until the next boot; we prefer the
|
|
92
|
+
# immediate move so a rolling deploy recovers work without a restart.
|
|
84
93
|
def bulk_requeue(in_progress)
|
|
85
94
|
return if in_progress.nil? || in_progress.empty?
|
|
86
95
|
|
|
87
|
-
|
|
88
|
-
config.redis do |conn|
|
|
89
|
-
conn.pipelined do |pipe|
|
|
90
|
-
grouped.each do |public_q, uows|
|
|
91
|
-
pipe.call('RPUSH', public_q, *uows.map(&:job))
|
|
92
|
-
end
|
|
93
|
-
end
|
|
94
|
-
end
|
|
96
|
+
config.redis { |conn| requeue_pipelined(conn, in_progress) }
|
|
95
97
|
end
|
|
96
98
|
|
|
97
99
|
# Prefixed queue keys (`queue:<name>`) in fetch order. Strict mode
|
|
@@ -107,12 +109,40 @@ module Wurk
|
|
|
107
109
|
names.map { |q| "#{Keys::QUEUE_PREFIX}#{q}" }
|
|
108
110
|
end
|
|
109
111
|
|
|
112
|
+
# Quiet hook (Manager#quiet). Flips the drain flag so retrieve_work
|
|
113
|
+
# short-circuits: once quieted, no processor can pull a fresh UoW, even
|
|
114
|
+
# one sitting in the between-jobs window (Processor#run only re-checks its
|
|
115
|
+
# own @done between iterations). Quiet is one-way — matches Sidekiq TSTP
|
|
116
|
+
# (spec §21.3), there is no un-terminate.
|
|
110
117
|
def terminate
|
|
111
118
|
@done = true
|
|
112
119
|
end
|
|
113
120
|
|
|
114
121
|
private
|
|
115
122
|
|
|
123
|
+
# One pipelined RELIABLE_REQUEUE EVALSHA per UoW. Mirrors
|
|
124
|
+
# Client#push_batched_pipelined: a pipelined EVALSHA surfaces NOSCRIPT
|
|
125
|
+
# only at finalize (never to eval_cached's inline rescue). Every command
|
|
126
|
+
# here is the same script, so a flushed cache fails all of them and
|
|
127
|
+
# applies none — recover by reloading once and replaying the whole
|
|
128
|
+
# pipeline via source-embedded EVAL.
|
|
129
|
+
def requeue_pipelined(conn, in_progress, eval_method: :eval_cached)
|
|
130
|
+
conn.pipelined do |pipe|
|
|
131
|
+
in_progress.each do |uow|
|
|
132
|
+
Wurk::Lua::Loader.public_send(
|
|
133
|
+
eval_method, pipe, :reliable_requeue,
|
|
134
|
+
keys: [self.class.private_queue_name(uow.queue), uow.queue],
|
|
135
|
+
argv: [uow.job]
|
|
136
|
+
)
|
|
137
|
+
end
|
|
138
|
+
end
|
|
139
|
+
rescue RedisClient::CommandError => e
|
|
140
|
+
raise unless e.message.to_s.start_with?('NOSCRIPT')
|
|
141
|
+
|
|
142
|
+
Wurk::Lua::Loader.script_load_all(conn)
|
|
143
|
+
requeue_pipelined(conn, in_progress, eval_method: :eval_with_source)
|
|
144
|
+
end
|
|
145
|
+
|
|
116
146
|
# SMEMBERS of the `paused` SET. One round-trip per fetch pass; the
|
|
117
147
|
# set is tiny in practice (one entry per paused queue) so the cost
|
|
118
148
|
# is dominated by the BLMOVE that follows. Returns a Set for O(1)
|
|
@@ -130,9 +160,12 @@ module Wurk
|
|
|
130
160
|
def blmove(public_q)
|
|
131
161
|
priv = self.class.private_queue_name(public_q)
|
|
132
162
|
timeout = poll_interval
|
|
133
|
-
#
|
|
134
|
-
#
|
|
135
|
-
|
|
163
|
+
# Dedicated fetch pool, not the main one: a parked BLMOVE holds its slot
|
|
164
|
+
# for the whole block window, so routing it here keeps idle fetchers from
|
|
165
|
+
# starving the main pool's background loops (#101). Extend the socket
|
|
166
|
+
# read-timeout one second past BLMOVE's own server-side timeout so the
|
|
167
|
+
# connection's read timeout can't fire while BLMOVE is legitimately blocked.
|
|
168
|
+
job = config.fetch_redis do |conn|
|
|
136
169
|
conn.blocking_call(timeout + 1, 'BLMOVE', public_q, priv, 'RIGHT', 'LEFT', timeout)
|
|
137
170
|
end
|
|
138
171
|
job ? UnitOfWork.new(queue: public_q, job: job, config: config) : nil
|
data/lib/wurk/fetcher.rb
CHANGED
|
@@ -7,5 +7,10 @@ module Wurk
|
|
|
7
7
|
class Fetcher
|
|
8
8
|
def retrieve_work; end
|
|
9
9
|
def bulk_requeue(in_progress); end
|
|
10
|
+
|
|
11
|
+
# Quiet hook: Manager#quiet calls this so retrieve_work can short-circuit
|
|
12
|
+
# and stop pulling new work the instant a process is quieted. No-op in the
|
|
13
|
+
# abstract base; Reliable flips its drain flag.
|
|
14
|
+
def terminate; end
|
|
10
15
|
end
|
|
11
16
|
end
|
data/lib/wurk/health.rb
CHANGED
|
@@ -28,40 +28,41 @@ module Wurk
|
|
|
28
28
|
# start/stop; safe to call from Launcher#run / Launcher#stop.
|
|
29
29
|
class Server
|
|
30
30
|
ACCEPT_TIMEOUT = 0.2
|
|
31
|
+
# How often a non-owner child re-attempts the shared port. Short enough
|
|
32
|
+
# that probes come back quickly after the owner dies, long enough not to
|
|
33
|
+
# spin. See #start_retry_loop.
|
|
34
|
+
RETRY_INTERVAL = 5
|
|
31
35
|
|
|
32
36
|
attr_reader :port, :bind
|
|
33
37
|
|
|
34
|
-
def initialize(launcher, port: DEFAULT_PORT, bind: DEFAULT_BIND,
|
|
35
|
-
|
|
36
|
-
@
|
|
37
|
-
@
|
|
38
|
-
@
|
|
39
|
-
@
|
|
40
|
-
@
|
|
41
|
-
@
|
|
42
|
-
@
|
|
38
|
+
def initialize(launcher, port: DEFAULT_PORT, bind: DEFAULT_BIND,
|
|
39
|
+
ready_window: DEFAULT_READY_WINDOW, retry_interval: RETRY_INTERVAL)
|
|
40
|
+
@launcher = launcher
|
|
41
|
+
@config = launcher.instance_variable_get(:@config)
|
|
42
|
+
@port = port
|
|
43
|
+
@bind = bind
|
|
44
|
+
@ready_window = ready_window
|
|
45
|
+
@retry_interval = retry_interval
|
|
46
|
+
@server = nil
|
|
47
|
+
@thread = nil
|
|
48
|
+
@retry_thread = nil
|
|
49
|
+
@done = false
|
|
43
50
|
end
|
|
44
51
|
|
|
45
52
|
def start
|
|
46
|
-
|
|
47
|
-
#
|
|
48
|
-
#
|
|
49
|
-
|
|
53
|
+
# Idempotent (see class doc): a second start on a live instance would
|
|
54
|
+
# re-bind the same port, hit EADDRINUSE, and null out @server/@thread —
|
|
55
|
+
# leaking the original listener so stop could never close it.
|
|
56
|
+
return self if running? || retrying?
|
|
57
|
+
|
|
50
58
|
@done = false
|
|
51
|
-
|
|
52
|
-
@thread.name = 'wurk-health'
|
|
53
|
-
self
|
|
54
|
-
rescue ::Errno::EADDRINUSE => e
|
|
55
|
-
# Swarm children all try to bind the same port — only the first wins.
|
|
56
|
-
# Don't crash the worker; just log and skip.
|
|
57
|
-
logger&.warn { "Wurk::Health: port #{@port} in use; health server NOT started (#{e.message})" }
|
|
58
|
-
@server = nil
|
|
59
|
-
@thread = nil
|
|
59
|
+
bind_and_serve || start_retry_loop
|
|
60
60
|
self
|
|
61
61
|
end
|
|
62
62
|
|
|
63
63
|
def stop
|
|
64
64
|
@done = true
|
|
65
|
+
stop_retry_loop
|
|
65
66
|
srv = @server
|
|
66
67
|
@server = nil
|
|
67
68
|
srv&.close
|
|
@@ -73,8 +74,59 @@ module Wurk
|
|
|
73
74
|
@thread&.alive? == true
|
|
74
75
|
end
|
|
75
76
|
|
|
77
|
+
# True while a non-owner child is still polling to take the shared port
|
|
78
|
+
# over (see #start_retry_loop). Distinct from #running?, which reports the
|
|
79
|
+
# accept thread specifically.
|
|
80
|
+
def retrying?
|
|
81
|
+
@retry_thread&.alive? == true
|
|
82
|
+
end
|
|
83
|
+
|
|
76
84
|
private
|
|
77
85
|
|
|
86
|
+
# One bind attempt. On success spins the accept thread and returns true.
|
|
87
|
+
# On EADDRINUSE — a sibling swarm child already owns the shared port —
|
|
88
|
+
# returns false so the caller schedules a retry.
|
|
89
|
+
def bind_and_serve
|
|
90
|
+
@server = ::TCPServer.new(@bind, @port)
|
|
91
|
+
# Capture the OS-assigned port when caller passed 0 (test pattern,
|
|
92
|
+
# also lets the kernel pick a free port at boot).
|
|
93
|
+
@port = @server.addr[1]
|
|
94
|
+
@thread = ::Thread.new { run }
|
|
95
|
+
@thread.name = 'wurk-health'
|
|
96
|
+
true
|
|
97
|
+
rescue ::Errno::EADDRINUSE
|
|
98
|
+
@server = nil
|
|
99
|
+
@thread = nil
|
|
100
|
+
false
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
# Non-owner children poll the shared port instead of giving up. A single
|
|
104
|
+
# bind-at-boot went dark to k8s the moment the owning child died —
|
|
105
|
+
# nothing rebound until the pod restarted. Now a survivor takes the port
|
|
106
|
+
# over within RETRY_INTERVAL of the owner's exit, so liveness/readiness
|
|
107
|
+
# ride out ordinary child churn (crash-respawn, rolling restart, recycle).
|
|
108
|
+
def start_retry_loop
|
|
109
|
+
logger&.warn do
|
|
110
|
+
"Wurk::Health: port #{@port} in use; polling every #{@retry_interval}s to take it over"
|
|
111
|
+
end
|
|
112
|
+
@retry_thread = ::Thread.new do
|
|
113
|
+
::Thread.current.name = 'wurk-health-retry'
|
|
114
|
+
::Thread.current.report_on_exception = false
|
|
115
|
+
sleep @retry_interval until @done || bind_and_serve
|
|
116
|
+
end
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
def stop_retry_loop
|
|
120
|
+
thread = @retry_thread
|
|
121
|
+
@retry_thread = nil
|
|
122
|
+
return unless thread
|
|
123
|
+
|
|
124
|
+
thread.wakeup if thread.alive?
|
|
125
|
+
thread.join(@retry_interval + 1)
|
|
126
|
+
rescue ThreadError
|
|
127
|
+
nil
|
|
128
|
+
end
|
|
129
|
+
|
|
78
130
|
def run
|
|
79
131
|
until @done
|
|
80
132
|
ready = ::IO.select([@server], nil, nil, ACCEPT_TIMEOUT)
|
data/lib/wurk/history.rb
CHANGED
|
@@ -4,6 +4,7 @@ require_relative 'component'
|
|
|
4
4
|
require_relative 'keys'
|
|
5
5
|
require_relative 'stats'
|
|
6
6
|
require_relative 'metrics/statsd'
|
|
7
|
+
require_relative 'timer_loop'
|
|
7
8
|
|
|
8
9
|
module Wurk
|
|
9
10
|
# Sidekiq Enterprise §5 Historical Metrics snapshotter. A leader-gated
|
|
@@ -63,30 +64,18 @@ module Wurk
|
|
|
63
64
|
|
|
64
65
|
def initialize(config)
|
|
65
66
|
@config = config
|
|
66
|
-
@interval = config.history_interval
|
|
67
67
|
@collector = config.history_collector
|
|
68
68
|
@stream_cap = config[:history_stream_cap] || STREAM_CAP
|
|
69
|
-
@
|
|
70
|
-
@mutex = ::Mutex.new
|
|
71
|
-
@sleeper = ::ConditionVariable.new
|
|
69
|
+
@timer = TimerLoop.new(config.history_interval)
|
|
72
70
|
@thread = nil
|
|
73
71
|
end
|
|
74
72
|
|
|
75
73
|
def start
|
|
76
|
-
@thread ||= safe_thread('history-snapshot')
|
|
77
|
-
wait
|
|
78
|
-
until @done
|
|
79
|
-
tick
|
|
80
|
-
wait
|
|
81
|
-
end
|
|
82
|
-
end
|
|
74
|
+
@thread ||= safe_thread('history-snapshot') { @timer.run { tick } } # rubocop:disable Naming/MemoizedInstanceVariableName
|
|
83
75
|
end
|
|
84
76
|
|
|
85
77
|
def terminate
|
|
86
|
-
@
|
|
87
|
-
@done = true
|
|
88
|
-
@sleeper.signal
|
|
89
|
-
end
|
|
78
|
+
@timer.terminate
|
|
90
79
|
end
|
|
91
80
|
|
|
92
81
|
# Leader-gated: only the elected leader emits, so N workers don't each
|
|
@@ -164,11 +153,5 @@ module Wurk
|
|
|
164
153
|
|
|
165
154
|
values.each { |field, value| client.gauge("sidekiq.#{field}", value) }
|
|
166
155
|
end
|
|
167
|
-
|
|
168
|
-
def wait
|
|
169
|
-
@mutex.synchronize do
|
|
170
|
-
@sleeper.wait(@mutex, @interval) unless @done
|
|
171
|
-
end
|
|
172
|
-
end
|
|
173
156
|
end
|
|
174
157
|
end
|
data/lib/wurk/job_retry.rb
CHANGED
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
require 'zlib'
|
|
4
4
|
require_relative 'component'
|
|
5
|
+
require_relative 'dead_set'
|
|
5
6
|
|
|
6
7
|
module Wurk
|
|
7
8
|
# Owns the retry pipeline. When perform raises, JobRetry decides whether to
|
|
@@ -194,7 +195,7 @@ module Wurk
|
|
|
194
195
|
if item.is_a?(::Float)
|
|
195
196
|
::Time.at(item)
|
|
196
197
|
else
|
|
197
|
-
::Time.at(item / 1000, item % 1000)
|
|
198
|
+
::Time.at(item / 1000, item % 1000, :millisecond)
|
|
198
199
|
end
|
|
199
200
|
end
|
|
200
201
|
|
|
@@ -262,10 +263,7 @@ module Wurk
|
|
|
262
263
|
|
|
263
264
|
def send_to_morgue(msg)
|
|
264
265
|
logger.info { "Adding dead #{msg['class']} job #{msg['jid']}" }
|
|
265
|
-
|
|
266
|
-
now = ::Time.now.to_f
|
|
267
|
-
redis { |conn| conn.call('ZADD', Keys::DEAD, now.to_s, payload) }
|
|
268
|
-
DeadSet.new.trim
|
|
266
|
+
DeadSet.new.kill_raw(Wurk.dump_json(msg))
|
|
269
267
|
end
|
|
270
268
|
|
|
271
269
|
def run_death_handlers(job, exception)
|
data/lib/wurk/launcher.rb
CHANGED
|
@@ -65,6 +65,7 @@ module Wurk
|
|
|
65
65
|
@started_at = nil
|
|
66
66
|
@heartbeat = nil
|
|
67
67
|
@heartbeat_thread = nil
|
|
68
|
+
@boot_reclaim_thread = nil
|
|
68
69
|
@health_server = build_health_server
|
|
69
70
|
end
|
|
70
71
|
|
|
@@ -94,7 +95,9 @@ module Wurk
|
|
|
94
95
|
@history&.start
|
|
95
96
|
@managers.each(&:start)
|
|
96
97
|
@reaper.start
|
|
97
|
-
|
|
98
|
+
# Run on a background thread so /ready probe isn't delayed by a large
|
|
99
|
+
# orphan sweep (reaper.reclaim! is atomic, but can scan many entries).
|
|
100
|
+
@boot_reclaim_thread = safe_thread('boot-reclaim', &method(:boot_reclaim))
|
|
98
101
|
@health_server&.start
|
|
99
102
|
end
|
|
100
103
|
|
|
@@ -231,6 +234,9 @@ module Wurk
|
|
|
231
234
|
@stopped = true
|
|
232
235
|
thread = @heartbeat_thread
|
|
233
236
|
return unless thread
|
|
237
|
+
# Embedded dashboard-TERM runs `stop` from the beat itself; a self-join
|
|
238
|
+
# raises ThreadError. @stopped is set, so the loop exits after this beat.
|
|
239
|
+
return if thread == Thread.current
|
|
234
240
|
|
|
235
241
|
begin
|
|
236
242
|
thread.wakeup
|
|
@@ -253,15 +259,32 @@ module Wurk
|
|
|
253
259
|
logger.info('Heartbeat stopping...')
|
|
254
260
|
end
|
|
255
261
|
|
|
262
|
+
# Dashboard-queued signals must behave exactly like OS signals, so a
|
|
263
|
+
# standalone process re-delivers to itself and lets the installed trap
|
|
264
|
+
# run — that wakes the main thread (CLI self-pipe / child dispatcher)
|
|
265
|
+
# so the process actually exits instead of stopping its managers and
|
|
266
|
+
# then parking forever. Embedded mode owns no traps (and self-TERM
|
|
267
|
+
# would kill the host app), so it calls quiet/stop directly — stop on
|
|
268
|
+
# its own thread because `stop` joins the heartbeat thread we're on.
|
|
256
269
|
def dispatch_signal(sig)
|
|
257
270
|
case sig
|
|
258
|
-
when 'TSTP'
|
|
259
|
-
|
|
271
|
+
when 'TSTP', 'TERM'
|
|
272
|
+
if @embedded
|
|
273
|
+
sig == 'TSTP' ? quiet : Thread.new { stop }
|
|
274
|
+
else
|
|
275
|
+
redeliver(sig)
|
|
276
|
+
end
|
|
260
277
|
else
|
|
261
278
|
logger.warn { "Unknown signal in #{identity}-signals: #{sig.inspect}" }
|
|
262
279
|
end
|
|
263
280
|
end
|
|
264
281
|
|
|
282
|
+
# Separate method so tests can stub it — really sending TERM/TSTP would
|
|
283
|
+
# kill or suspend the test process.
|
|
284
|
+
def redeliver(sig)
|
|
285
|
+
::Process.kill(sig, ::Process.pid)
|
|
286
|
+
end
|
|
287
|
+
|
|
265
288
|
def build_poller
|
|
266
289
|
Wurk::Scheduled::Poller.new(@config)
|
|
267
290
|
end
|
|
@@ -1,14 +1,28 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require_relative '../job_retry'
|
|
4
|
+
|
|
3
5
|
module Wurk
|
|
4
6
|
module Limiter
|
|
7
|
+
# Control signal raised after a job has been rescheduled: the limiter has
|
|
8
|
+
# already re-enqueued it (via `Client.push` at `Time.now + backoff`), so
|
|
9
|
+
# the run is *neither* a success nor a failure. Subclassing
|
|
10
|
+
# `JobRetry::Skip` routes it through the existing "middleware re-pushed the
|
|
11
|
+
# job — ack cleanly, book no retry" contract: the retrier re-raises it
|
|
12
|
+
# untouched, the outer Batch middleware skips both acks (its `rescue
|
|
13
|
+
# Handled` re-raises), and the Processor acks the UnitOfWork. Returning
|
|
14
|
+
# normally instead would let the outer Batch onion ack success for a job
|
|
15
|
+
# that never ran. Internal only — not part of the Sidekiq drop-in surface.
|
|
16
|
+
class Rescheduled < Wurk::JobRetry::Skip; end
|
|
17
|
+
|
|
5
18
|
# Catches OverLimit (and any class registered in `Limiter.config.errors`),
|
|
6
19
|
# bumps `job['overrated']`, and decides what to do next:
|
|
7
20
|
#
|
|
8
21
|
# * reschedule disabled (`reschedule: 0`) → re-raise so the normal
|
|
9
22
|
# retry/dead pipeline handles it (spec §1.2/§1.4 behaviour).
|
|
10
23
|
# * still under the cap → reschedule onto the same queue at
|
|
11
|
-
# `Time.now + backoff` via `Client.push
|
|
24
|
+
# `Time.now + backoff` via `Client.push`, then raise `Rescheduled` so
|
|
25
|
+
# the outcome is neither success nor failure (batch onion skips acks).
|
|
12
26
|
# * cap reached (`overrated >= reschedule`, default 20) → **poison
|
|
13
27
|
# brake** (#16): a job that's still rate-limited after N reschedules
|
|
14
28
|
# is saturating the limiter, so instead of dumping it into another
|
|
@@ -58,6 +72,7 @@ module Wurk
|
|
|
58
72
|
backoff_proc = (limiter && limiter.options[:backoff]) || Wurk::Limiter.config.backoff
|
|
59
73
|
delay = backoff_proc.call(limiter, job, exc).to_f
|
|
60
74
|
Wurk::Client.new.push(job.merge('at' => ::Time.now.to_f + delay))
|
|
75
|
+
raise Rescheduled
|
|
61
76
|
end
|
|
62
77
|
|
|
63
78
|
# Poison brake: stamp a clear reason, drop the job in the dead set, and
|
data/lib/wurk/limiter.rb
CHANGED
|
@@ -77,7 +77,7 @@ module Wurk
|
|
|
77
77
|
@backoff = DEFAULT_BACKOFF
|
|
78
78
|
@errors = [OverLimit]
|
|
79
79
|
@redis = nil
|
|
80
|
-
@
|
|
80
|
+
@pool = nil
|
|
81
81
|
end
|
|
82
82
|
|
|
83
83
|
# Accept either a Hash (the documented Sidekiq Ent shape — `{ size:,
|
|
@@ -86,7 +86,9 @@ module Wurk
|
|
|
86
86
|
# responsibility (same contract as Wurk.redis_pool).
|
|
87
87
|
def redis=(value)
|
|
88
88
|
@redis = value
|
|
89
|
-
|
|
89
|
+
# Reset the memo `pool` actually reads — resetting a different ivar
|
|
90
|
+
# left the old pool pinned after `config.redis = {...}`.
|
|
91
|
+
@pool = nil
|
|
90
92
|
end
|
|
91
93
|
|
|
92
94
|
def pool
|
|
@@ -97,9 +99,8 @@ module Wurk
|
|
|
97
99
|
when Hash
|
|
98
100
|
Wurk::RedisPool.new(
|
|
99
101
|
size: @redis[:size] || 10,
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
name: 'limiter'
|
|
102
|
+
name: 'limiter',
|
|
103
|
+
**@redis.except(:size, :name)
|
|
103
104
|
)
|
|
104
105
|
else
|
|
105
106
|
raise ArgumentError, "Limiter.config.redis must be Hash or RedisPool, got #{@redis.class}"
|
data/lib/wurk/lua/loader.rb
CHANGED
|
@@ -14,12 +14,18 @@ module Wurk
|
|
|
14
14
|
NOSCRIPT_PREFIX = 'NOSCRIPT'
|
|
15
15
|
|
|
16
16
|
class << self
|
|
17
|
-
# Eagerly upload every registered script to the given connection
|
|
18
|
-
#
|
|
19
|
-
#
|
|
20
|
-
#
|
|
17
|
+
# Eagerly upload every registered script to the given connection in a
|
|
18
|
+
# single pipelined round-trip (all SCRIPT LOADs, one RTT — not one per
|
|
19
|
+
# script). Idempotent on the Redis side: `SCRIPT LOAD` of the same source
|
|
20
|
+
# returns the same SHA no matter how often it runs. ChildBoot calls this
|
|
21
|
+
# once per child right after the post-fork reconnect so the first real
|
|
22
|
+
# EVALSHA hits a warm cache instead of paying a NOSCRIPT reload. Transient
|
|
23
|
+
# connection errors are the pool wrapper's job (Wurk::RedisPool#with);
|
|
24
|
+
# this only ships the loads.
|
|
21
25
|
def script_load_all(redis)
|
|
22
|
-
|
|
26
|
+
redis.pipelined do |pipe|
|
|
27
|
+
SCRIPTS.each_value { |src| pipe.call('SCRIPT', 'LOAD', src) }
|
|
28
|
+
end
|
|
23
29
|
end
|
|
24
30
|
|
|
25
31
|
# @param redis [RedisClient] a single connection (not a pool)
|
data/lib/wurk/lua.rb
CHANGED
|
@@ -37,28 +37,91 @@ module Wurk
|
|
|
37
37
|
# Pro reliable scheduler: atomically promote all due jobs in a sorted
|
|
38
38
|
# set to their target queues. Pure-Ruby promotion does ZRANGE → ZREM →
|
|
39
39
|
# LPUSH non-atomically and can lose jobs on a mid-step crash.
|
|
40
|
+
#
|
|
41
|
+
# Each promoted payload is restamped with a fresh `enqueued_at` (ARGV[3],
|
|
42
|
+
# epoch ms): that field marks arrival on an *immediate* queue, so a job
|
|
43
|
+
# leaving `schedule`/`retry` must get a new one at promotion rather than
|
|
44
|
+
# keep its stale scheduled-origin value (or none). This matches the default
|
|
45
|
+
# Ruby scheduler, whose push path restamps `enqueued_at` too
|
|
46
|
+
# (Client#push_plain / #push_batched) — so both schedulers emit
|
|
47
|
+
# wire-identical promoted payloads (spec §7.1).
|
|
48
|
+
#
|
|
49
|
+
# The restamp is a surgical string patch, NOT a cjson.decode -> cjson.encode
|
|
50
|
+
# round-trip: cjson maps every JSON number to a Lua double, so re-encoding
|
|
51
|
+
# would silently corrupt integer args past 2^53 (snowflake IDs, 64-bit
|
|
52
|
+
# counters) and reformat 15+ digit numbers into scientific notation --
|
|
53
|
+
# breaking wire-compat AND diverging from the loss-free Ruby scheduler,
|
|
54
|
+
# whose Ruby-side JSON keeps big integers exact. So the stored member is
|
|
55
|
+
# preserved byte-for-byte and only its top-level `enqueued_at` is rewritten:
|
|
56
|
+
# replaced in place when present (retry members keep theirs), inserted after
|
|
57
|
+
# the opening brace when absent (schedule members are stored stripped of it,
|
|
58
|
+
# via Client#push_scheduled). cjson.decode is still called -- but only to
|
|
59
|
+
# read the `queue` string and test for a top-level `enqueued_at`, never to
|
|
60
|
+
# re-serialize, so a lossy decode never reaches the payload. (Residual: a
|
|
61
|
+
# retry member whose user args embed a numeric key literally named
|
|
62
|
+
# `enqueued_at` ahead of the top-level one patches the arg instead --
|
|
63
|
+
# vanishingly rare, and still strictly safer than round-tripping every arg.)
|
|
40
64
|
# KEYS = [sorted_set, queues_set]
|
|
41
|
-
# ARGV = [now, queue_prefix]
|
|
65
|
+
# ARGV = [now, queue_prefix, now_ms]
|
|
42
66
|
# Returns the number of jobs promoted.
|
|
43
67
|
# Order matters: decode + push BEFORE zrem. Redis Lua has no rollback,
|
|
44
68
|
# so a failed cjson.decode after a zrem would lose the job. Decode first;
|
|
45
|
-
# push first; only then remove from the sorted set
|
|
46
|
-
#
|
|
69
|
+
# push first; only then remove from the sorted set — and zrem the ORIGINAL
|
|
70
|
+
# member, not the restamped copy. Worst case is a crash between lpush and
|
|
71
|
+
# zrem → at-least-once redelivery, never loss.
|
|
72
|
+
# ARGV[4] caps members per call: Lua is atomic and single-threaded in
|
|
73
|
+
# Redis, so an unbatched promote of a post-outage backlog (100k+ due
|
|
74
|
+
# members) would block every client for the whole sweep. The Ruby caller
|
|
75
|
+
# loops until a short batch comes back.
|
|
47
76
|
RELIABLE_SCHEDULE_PROMOTE = <<~LUA
|
|
48
|
-
local jobs = redis.call("zrangebyscore", KEYS[1], "-inf", ARGV[1])
|
|
77
|
+
local jobs = redis.call("zrangebyscore", KEYS[1], "-inf", ARGV[1], "LIMIT", 0, tonumber(ARGV[4]))
|
|
49
78
|
for i = 1, #jobs do
|
|
50
79
|
local job = jobs[i]
|
|
51
|
-
local
|
|
80
|
+
local decoded = cjson.decode(job)
|
|
81
|
+
local q = decoded["queue"]
|
|
82
|
+
local stamped
|
|
83
|
+
if decoded["enqueued_at"] == nil then
|
|
84
|
+
stamped = string.gsub(job, "^{", '{"enqueued_at":' .. ARGV[3] .. ",", 1)
|
|
85
|
+
else
|
|
86
|
+
stamped = string.gsub(job, '"enqueued_at":%-?%d[%d.eE+-]*', '"enqueued_at":' .. ARGV[3], 1)
|
|
87
|
+
end
|
|
52
88
|
redis.call("sadd", KEYS[2], q)
|
|
53
|
-
redis.call("lpush", ARGV[2] .. q,
|
|
89
|
+
redis.call("lpush", ARGV[2] .. q, stamped)
|
|
54
90
|
redis.call("zrem", KEYS[1], job)
|
|
55
91
|
end
|
|
56
92
|
return #jobs
|
|
57
93
|
LUA
|
|
58
94
|
|
|
95
|
+
# Reliable fetch (Pro super_fetch §3) shutdown requeue: atomically move
|
|
96
|
+
# one in-flight job from a per-process private list back to its public
|
|
97
|
+
# queue. The LREM guard is the whole point — RPUSH runs only when the job
|
|
98
|
+
# was still in the private list (LREM removed exactly 1). A job the
|
|
99
|
+
# Processor ACKed in the window between hard_shutdown's cross-thread `job`
|
|
100
|
+
# read and this move (LREM removes 0) is NOT re-pushed, so the job lands in
|
|
101
|
+
# exactly one place and can't double-execute. RPUSH (public tail) not LPUSH
|
|
102
|
+
# so the reclaimed job is fetched next — LMOVE pops the tail — ahead of
|
|
103
|
+
# fresh LPUSH'd enqueues.
|
|
104
|
+
# KEYS = [private_list, public_queue]
|
|
105
|
+
# ARGV = [job_json]
|
|
106
|
+
# Returns 1 when the job was moved, 0 when it was already acked.
|
|
107
|
+
RELIABLE_REQUEUE = <<~LUA
|
|
108
|
+
if redis.call("lrem", KEYS[1], 1, ARGV[1]) == 1 then
|
|
109
|
+
redis.call("rpush", KEYS[2], ARGV[1])
|
|
110
|
+
return 1
|
|
111
|
+
end
|
|
112
|
+
return 0
|
|
113
|
+
LUA
|
|
114
|
+
|
|
59
115
|
# Pro Batch: register a job into a batch and push it to its queue
|
|
60
116
|
# atomically. Keeps total/pending in sync with the jids set.
|
|
61
117
|
#
|
|
118
|
+
# SADD into the live jids set is the registration guard: total/pending
|
|
119
|
+
# increment only when the jid is genuinely new (SADD == 1). A jid already
|
|
120
|
+
# live is a re-push — a retry or scheduled promotion re-enqueueing a job
|
|
121
|
+
# that never left the batch — so it re-LPUSHes the payload but must NOT
|
|
122
|
+
# recount, or pending inflates past the acks and `:success` never fires
|
|
123
|
+
# (spec §2.3/§2.5: total = distinct jobs added, pending = not-yet-succeeded).
|
|
124
|
+
#
|
|
62
125
|
# A jid found in `b-<bid>-died` is a manual retry of a dead job (morgue
|
|
63
126
|
# "retry" / "add to queue") — it rejoins the live set without recounting:
|
|
64
127
|
# total and pending already include it, because a death never decrements
|
|
@@ -79,15 +142,41 @@ module Wurk
|
|
|
79
142
|
redis.call("zrem", KEYS[6], ARGV[4])
|
|
80
143
|
end
|
|
81
144
|
else
|
|
82
|
-
redis.call("
|
|
83
|
-
|
|
84
|
-
|
|
145
|
+
if redis.call("sadd", KEYS[2], ARGV[2]) == 1 then
|
|
146
|
+
redis.call("hincrby", KEYS[1], "total", 1)
|
|
147
|
+
redis.call("hincrby", KEYS[1], "pending", 1)
|
|
148
|
+
end
|
|
85
149
|
end
|
|
86
150
|
redis.call("sadd", KEYS[4], ARGV[1])
|
|
87
151
|
redis.call("lpush", KEYS[3], ARGV[3])
|
|
88
152
|
return 1
|
|
89
153
|
LUA
|
|
90
154
|
|
|
155
|
+
# Pro Batch: register a scheduled (`at`) job into a batch AND ZADD it onto
|
|
156
|
+
# the `schedule` set, atomically. This is the deferred sibling of BATCH_PUSH:
|
|
157
|
+
# same SADD-guarded total/pending counting, but the enqueue action is a ZADD
|
|
158
|
+
# (schedule for later) instead of an LPUSH (queue now). A `perform_in` inside
|
|
159
|
+
# `batch.jobs` must move `total`/`pending` at *creation* — otherwise the
|
|
160
|
+
# empty-marker check (batch.rb) sees no counter movement and fires
|
|
161
|
+
# `:complete`/`:success` while real jobs still sit in `schedule`.
|
|
162
|
+
#
|
|
163
|
+
# No died / dead-batches handling (unlike BATCH_PUSH): a job scheduled at
|
|
164
|
+
# creation time is always new to the batch. When the scheduler later promotes
|
|
165
|
+
# it, the re-push routes through BATCH_PUSH, whose guard finds the jid already
|
|
166
|
+
# live (SADD == 0) → pure LPUSH, no recount. So registration happens exactly
|
|
167
|
+
# once, here, at enqueue.
|
|
168
|
+
# KEYS = [schedule, b-<bid>, b-<bid>-jids]
|
|
169
|
+
# ARGV = [at_score, job_json, jid]
|
|
170
|
+
# Returns 1.
|
|
171
|
+
BATCH_SCHEDULE = <<~LUA
|
|
172
|
+
if redis.call("sadd", KEYS[3], ARGV[3]) == 1 then
|
|
173
|
+
redis.call("hincrby", KEYS[2], "total", 1)
|
|
174
|
+
redis.call("hincrby", KEYS[2], "pending", 1)
|
|
175
|
+
end
|
|
176
|
+
redis.call("zadd", KEYS[1], ARGV[1], ARGV[2])
|
|
177
|
+
return 1
|
|
178
|
+
LUA
|
|
179
|
+
|
|
91
180
|
# Pro Batch: ACK a job that completed successfully. SREM from the live
|
|
92
181
|
# jids set and decrement pending iff the jid was a member (idempotent
|
|
93
182
|
# against double-success on a flaky retry). A success also clears any
|
|
@@ -259,7 +348,9 @@ module Wurk
|
|
|
259
348
|
zpopbyscore: ZPOPBYSCORE,
|
|
260
349
|
bulk_push: BULK_PUSH,
|
|
261
350
|
reliable_schedule_promote: RELIABLE_SCHEDULE_PROMOTE,
|
|
351
|
+
reliable_requeue: RELIABLE_REQUEUE,
|
|
262
352
|
batch_push: BATCH_PUSH,
|
|
353
|
+
batch_schedule: BATCH_SCHEDULE,
|
|
263
354
|
batch_ack_success: BATCH_ACK_SUCCESS,
|
|
264
355
|
batch_ack_failed: BATCH_ACK_FAILED,
|
|
265
356
|
batch_ack_complete: BATCH_ACK_COMPLETE,
|