wurk 1.3.0 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +1 -0
- data/lib/wurk/batch/callbacks.rb +82 -12
- data/lib/wurk/batch/death_handler.rb +7 -4
- data/lib/wurk/batch/server_middleware.rb +1 -1
- data/lib/wurk/batch.rb +121 -15
- data/lib/wurk/capsule.rb +5 -4
- data/lib/wurk/cli.rb +48 -14
- data/lib/wurk/client/buffered.rb +193 -43
- data/lib/wurk/client.rb +87 -14
- data/lib/wurk/compat.rb +1 -1
- data/lib/wurk/component.rb +2 -2
- data/lib/wurk/configuration.rb +10 -2
- data/lib/wurk/cron.rb +94 -37
- data/lib/wurk/deploy.rb +5 -3
- data/lib/wurk/embedded.rb +13 -0
- data/lib/wurk/fetcher/reaper.rb +113 -56
- data/lib/wurk/fetcher/reliable.rb +62 -9
- data/lib/wurk/heartbeat.rb +22 -10
- data/lib/wurk/history.rb +13 -1
- data/lib/wurk/launcher.rb +133 -66
- data/lib/wurk/leader.rb +29 -10
- data/lib/wurk/limiter/base.rb +8 -10
- data/lib/wurk/limiter/bucket.rb +1 -1
- data/lib/wurk/limiter/concurrent.rb +27 -22
- data/lib/wurk/limiter/window.rb +13 -11
- data/lib/wurk/limiter.rb +7 -4
- data/lib/wurk/lua.rb +97 -14
- data/lib/wurk/manager.rb +29 -13
- data/lib/wurk/metrics/history.rb +4 -3
- data/lib/wurk/metrics/queue_rollup.rb +13 -1
- data/lib/wurk/metrics/rollup.rb +13 -1
- data/lib/wurk/middleware/interrupt_handler.rb +7 -6
- data/lib/wurk/middleware/poison_pill.rb +70 -29
- data/lib/wurk/middleware.rb +2 -2
- data/lib/wurk/pool_checkout.rb +29 -0
- data/lib/wurk/process_set.rb +10 -5
- data/lib/wurk/processor.rb +6 -0
- data/lib/wurk/profiler.rb +3 -2
- data/lib/wurk/queue.rb +10 -7
- data/lib/wurk/rails_boot.rb +38 -7
- data/lib/wurk/redis_client_adapter.rb +48 -4
- data/lib/wurk/redis_options.rb +142 -0
- data/lib/wurk/redis_pool.rb +102 -39
- data/lib/wurk/scheduled.rb +30 -2
- data/lib/wurk/stats.rb +14 -9
- data/lib/wurk/swarm/child_boot.rb +12 -0
- data/lib/wurk/swarm.rb +174 -33
- data/lib/wurk/timer_loop.rb +14 -0
- data/lib/wurk/version.rb +1 -1
- data/lib/wurk/web/enterprise.rb +58 -6
- data/lib/wurk/web/extension.rb +1 -1
- data/lib/wurk/web/search.rb +5 -3
- data/lib/wurk.rb +53 -2
- data/vendor/assets/dashboard/assets/{BatchDetail-YRymNsrB.js → BatchDetail-OmC5NPgw.js} +1 -1
- data/vendor/assets/dashboard/assets/{Batches-HY4hHdQU.js → Batches-CIpai7St.js} +1 -1
- data/vendor/assets/dashboard/assets/{Busy-FCEN1Bpx.js → Busy-A_kwSR6Q.js} +1 -1
- data/vendor/assets/dashboard/assets/{Cron-DO3J2zcp.js → Cron-BG7HTqlp.js} +1 -1
- data/vendor/assets/dashboard/assets/{Dashboard-B9rOrkzk.js → Dashboard-A_ToqHoo.js} +1 -1
- data/vendor/assets/dashboard/assets/{Dead-Bi4GGk9a.js → Dead-8J21jMyK.js} +1 -1
- data/vendor/assets/dashboard/assets/Extension-B4Q9FIQu.js +1 -0
- data/vendor/assets/dashboard/assets/{FilterBox-IJkHYpdm.js → FilterBox-Fh_Ae7UW.js} +1 -1
- data/vendor/assets/dashboard/assets/{JobDetailModal-DS1ypyoc.js → JobDetailModal-Ceng0PMB.js} +1 -1
- data/vendor/assets/dashboard/assets/{Limiters-Nz7UbNeJ.js → Limiters-CruDWvNZ.js} +1 -1
- data/vendor/assets/dashboard/assets/{Metrics-BBTDxcaE.js → Metrics-CIT7VCoN.js} +1 -1
- data/vendor/assets/dashboard/assets/Modal-CN3rdKA_.js +1 -0
- data/vendor/assets/dashboard/assets/{Queues-D9PH_THs.js → Queues-D86FYohJ.js} +1 -1
- data/vendor/assets/dashboard/assets/{Retries-CAKzDgYG.js → Retries-Bz1O1D-i.js} +1 -1
- data/vendor/assets/dashboard/assets/{Scheduled-DTYw1X8S.js → Scheduled-B6h2akTu.js} +1 -1
- data/vendor/assets/dashboard/assets/{Search-c4vFxDG_.js → Search-OOu22e5s.js} +1 -1
- data/vendor/assets/dashboard/assets/index-BdiUEDXX.css +1 -0
- data/vendor/assets/dashboard/assets/index-D_lSDwKw.js +141 -0
- data/vendor/assets/dashboard/assets/{useResetPageOnEmpty-B_FsMah6.js → useResetPageOnEmpty-dVPGEWzn.js} +1 -1
- data/vendor/assets/dashboard/index.html +2 -2
- data/vendor/assets/dashboard/wurk-manifest.json +2 -2
- metadata +22 -20
- data/vendor/assets/dashboard/assets/Extension-BSv8ddW_.js +0 -1
- data/vendor/assets/dashboard/assets/Modal-Crrsu64-.js +0 -1
- data/vendor/assets/dashboard/assets/index-BxjdeuOa.css +0 -1
- data/vendor/assets/dashboard/assets/index-DQu7WY9y.js +0 -141
data/lib/wurk/limiter.rb
CHANGED
|
@@ -4,6 +4,7 @@ require 'json'
|
|
|
4
4
|
require 'digest'
|
|
5
5
|
require 'securerandom'
|
|
6
6
|
require_relative 'lua'
|
|
7
|
+
require_relative 'pool_checkout'
|
|
7
8
|
|
|
8
9
|
module Wurk
|
|
9
10
|
# Sidekiq Enterprise rate limiters: concurrent, bucket, window, leaky,
|
|
@@ -24,7 +25,10 @@ module Wurk
|
|
|
24
25
|
# Layout (one file per type under `lib/wurk/limiter/`):
|
|
25
26
|
# * `Limiter::Base` owns the metadata write (lmtr:{name}) + the global
|
|
26
27
|
# `lmtr-list` registration so the Web UI can list every limiter, and
|
|
27
|
-
# the uniform `status` shape.
|
|
28
|
+
# the uniform `status` shape. Membership is added once per metadata
|
|
29
|
+
# write and swept back out by `Web::Enterprise::Limits.list` once the
|
|
30
|
+
# metadata expires — the two halves of keeping `lmtr-list` bounded
|
|
31
|
+
# under interpolated names.
|
|
28
32
|
# * Per-type subclasses (Concurrent / Bucket / Window / Leaky / Points)
|
|
29
33
|
# own their acquire/wait loop. Each delegates the atomic step to a
|
|
30
34
|
# Lua script in `lib/wurk/lua/limiter_*.lua`.
|
|
@@ -146,9 +150,8 @@ module Wurk
|
|
|
146
150
|
# Redis access: caller-supplied pool (Limiter.configure.redis = …) wins,
|
|
147
151
|
# else fall back to the default Wurk pool. This is the same hierarchy
|
|
148
152
|
# Sidekiq Ent documents — dedicated rate-limiter pool is opt-in.
|
|
149
|
-
def redis(&)
|
|
150
|
-
|
|
151
|
-
pool.with(&)
|
|
153
|
+
def redis(idempotent: false, &)
|
|
154
|
+
PoolCheckout.with(config.pool || Wurk.redis_pool, idempotent, &)
|
|
152
155
|
end
|
|
153
156
|
|
|
154
157
|
# `ZRANGE key 0 0 WITHSCORES` yields a single [member, score] pair, but the
|
data/lib/wurk/lua.rb
CHANGED
|
@@ -131,8 +131,16 @@ module Wurk
|
|
|
131
131
|
# success after the dead job is manually retried to success). The
|
|
132
132
|
# `b-<bid>-death` notify dedup key is untouched, so `:death` cannot
|
|
133
133
|
# re-fire.
|
|
134
|
+
#
|
|
135
|
+
# The two EXPIRE NX calls are what keep the batch bounded: `b-<bid>-jids` is
|
|
136
|
+
# born here, and only `Batch#ensure_first_flush!` ever stamped a TTL — on the
|
|
137
|
+
# hash alone — so an abandoned or invalidated batch leaked its live-jid set
|
|
138
|
+
# forever. A late re-push (retry, scheduled promotion) can also resurrect an
|
|
139
|
+
# already-expired `b-<bid>` through the HINCRBYs above, again TTL-less. NX
|
|
140
|
+
# stamps only keys that currently have no TTL, so a running batch's clock and
|
|
141
|
+
# the shorter post-success `linger` window (Callbacks#apply_linger) both win.
|
|
134
142
|
# KEYS = [b-<bid>, b-<bid>-jids, queue_list, queues_set, b-<bid>-died, dead-batches]
|
|
135
|
-
# ARGV = [queue_name, jid, job_json, bid]
|
|
143
|
+
# ARGV = [queue_name, jid, job_json, bid, expiry_seconds]
|
|
136
144
|
# Returns 1.
|
|
137
145
|
BATCH_PUSH = <<~LUA
|
|
138
146
|
if redis.call("srem", KEYS[5], ARGV[2]) == 1 then
|
|
@@ -147,6 +155,8 @@ module Wurk
|
|
|
147
155
|
redis.call("hincrby", KEYS[1], "pending", 1)
|
|
148
156
|
end
|
|
149
157
|
end
|
|
158
|
+
redis.call("expire", KEYS[1], ARGV[5], "NX")
|
|
159
|
+
redis.call("expire", KEYS[2], ARGV[5], "NX")
|
|
150
160
|
redis.call("sadd", KEYS[4], ARGV[1])
|
|
151
161
|
redis.call("lpush", KEYS[3], ARGV[3])
|
|
152
162
|
return 1
|
|
@@ -165,14 +175,19 @@ module Wurk
|
|
|
165
175
|
# it, the re-push routes through BATCH_PUSH, whose guard finds the jid already
|
|
166
176
|
# live (SADD == 0) → pure LPUSH, no recount. So registration happens exactly
|
|
167
177
|
# once, here, at enqueue.
|
|
178
|
+
#
|
|
179
|
+
# The other path that can create `b-<bid>-jids`, so it carries the same NX
|
|
180
|
+
# expiry stamp as BATCH_PUSH.
|
|
168
181
|
# KEYS = [schedule, b-<bid>, b-<bid>-jids]
|
|
169
|
-
# ARGV = [at_score, job_json, jid]
|
|
182
|
+
# ARGV = [at_score, job_json, jid, expiry_seconds]
|
|
170
183
|
# Returns 1.
|
|
171
184
|
BATCH_SCHEDULE = <<~LUA
|
|
172
185
|
if redis.call("sadd", KEYS[3], ARGV[3]) == 1 then
|
|
173
186
|
redis.call("hincrby", KEYS[2], "total", 1)
|
|
174
187
|
redis.call("hincrby", KEYS[2], "pending", 1)
|
|
175
188
|
end
|
|
189
|
+
redis.call("expire", KEYS[2], ARGV[4], "NX")
|
|
190
|
+
redis.call("expire", KEYS[3], ARGV[4], "NX")
|
|
176
191
|
redis.call("zadd", KEYS[1], ARGV[1], ARGV[2])
|
|
177
192
|
return 1
|
|
178
193
|
LUA
|
|
@@ -208,13 +223,18 @@ module Wurk
|
|
|
208
223
|
# currently in a failing/retrying state. Re-failures of the same jid are
|
|
209
224
|
# idempotent. Cleared by BATCH_ACK_SUCCESS (retry passed) or
|
|
210
225
|
# BATCH_ACK_COMPLETE (job died). Spec §2.5, §2.8.
|
|
226
|
+
#
|
|
227
|
+
# `b-<bid>-failed` is born here, so it carries the same NX expiry stamp as
|
|
228
|
+
# BATCH_PUSH.
|
|
211
229
|
# KEYS = [b-<bid>, b-<bid>-failed]
|
|
212
|
-
# ARGV = [jid]
|
|
230
|
+
# ARGV = [jid, expiry_seconds]
|
|
213
231
|
# Returns 1.
|
|
214
232
|
BATCH_ACK_FAILED = <<~LUA
|
|
215
233
|
if redis.call("sadd", KEYS[2], ARGV[1]) == 1 then
|
|
216
234
|
redis.call("hincrby", KEYS[1], "failures", 1)
|
|
217
235
|
end
|
|
236
|
+
redis.call("expire", KEYS[1], ARGV[2], "NX")
|
|
237
|
+
redis.call("expire", KEYS[2], ARGV[2], "NX")
|
|
218
238
|
return 1
|
|
219
239
|
LUA
|
|
220
240
|
|
|
@@ -224,8 +244,11 @@ module Wurk
|
|
|
224
244
|
# live jids so the batch can fire `:complete` even with terminally failed
|
|
225
245
|
# jobs. `b-<bid>-failed` holds only currently-retrying jids; `b-<bid>-died`
|
|
226
246
|
# holds terminally-dead ones (spec §2.8 — the two sets are distinct).
|
|
247
|
+
#
|
|
248
|
+
# `b-<bid>-died` is born here, so it carries the same NX expiry stamp as
|
|
249
|
+
# BATCH_PUSH.
|
|
227
250
|
# KEYS = [b-<bid>, b-<bid>-jids, b-<bid>-died, b-<bid>-failed]
|
|
228
|
-
# ARGV = [jid]
|
|
251
|
+
# ARGV = [jid, expiry_seconds]
|
|
229
252
|
# Returns [live_jids_remaining, died_count, first_death]. `first_death`
|
|
230
253
|
# is 1 the first time *any* jid is SADDed into the died set, 0 thereafter
|
|
231
254
|
# — caller uses it to fire `:death` exactly once per batch.
|
|
@@ -236,6 +259,8 @@ module Wurk
|
|
|
236
259
|
redis.call("hincrby", KEYS[1], "failures", -1)
|
|
237
260
|
end
|
|
238
261
|
local died_added = redis.call("sadd", KEYS[3], ARGV[1])
|
|
262
|
+
redis.call("expire", KEYS[1], ARGV[2], "NX")
|
|
263
|
+
redis.call("expire", KEYS[3], ARGV[2], "NX")
|
|
239
264
|
local first_death = 0
|
|
240
265
|
if was_pre_existing_death == 0 and died_added == 1 then
|
|
241
266
|
first_death = 1
|
|
@@ -262,10 +287,21 @@ module Wurk
|
|
|
262
287
|
# same reopened batch cannot lose each other's writes. Refuses to write
|
|
263
288
|
# when the batch hash is gone — resurrecting a bare hash would create a
|
|
264
289
|
# batch that can never fire anything.
|
|
290
|
+
#
|
|
291
|
+
# Appends are deduped and capped. Reopening the same batch inside every
|
|
292
|
+
# job and re-registering its callback is a common shape, and each append
|
|
293
|
+
# pays a decode + encode of the whole array: unbounded that is O(N^2) Lua
|
|
294
|
+
# work on the way in and N identical callback jobs at fire time. An
|
|
295
|
+
# identical triple is therefore a no-op — the registration is already
|
|
296
|
+
# satisfied by the entry that is there — and past the cap the append is
|
|
297
|
+
# refused rather than allowed to grow the hash field without limit.
|
|
298
|
+
# Both sides of the comparison are normalised through cjson so an entry
|
|
299
|
+
# written by Ruby's `to_json` at first flush matches one appended here.
|
|
265
300
|
# KEYS = [b-<bid>]
|
|
266
|
-
# ARGV = [callback triple JSON, event name]
|
|
267
|
-
# Returns -1 when the batch hash does not exist
|
|
268
|
-
# fired flag ("1", or nil when
|
|
301
|
+
# ARGV = [callback triple JSON, event name, max callbacks]
|
|
302
|
+
# Returns -1 when the batch hash does not exist and -2 when the cap
|
|
303
|
+
# refused the append; otherwise the event's fired flag ("1", or nil when
|
|
304
|
+
# it has not fired yet).
|
|
269
305
|
BATCH_APPEND_CALLBACK = <<~LUA
|
|
270
306
|
if redis.call("exists", KEYS[1]) == 0 then
|
|
271
307
|
return -1
|
|
@@ -277,19 +313,30 @@ module Wurk
|
|
|
277
313
|
else
|
|
278
314
|
list = {}
|
|
279
315
|
end
|
|
280
|
-
|
|
316
|
+
local entry = cjson.decode(ARGV[1])
|
|
317
|
+
local encoded = cjson.encode(entry)
|
|
318
|
+
for i = 1, #list do
|
|
319
|
+
if cjson.encode(list[i]) == encoded then
|
|
320
|
+
return redis.call("hget", KEYS[1], ARGV[2])
|
|
321
|
+
end
|
|
322
|
+
end
|
|
323
|
+
if #list >= tonumber(ARGV[3]) then
|
|
324
|
+
return -2
|
|
325
|
+
end
|
|
326
|
+
list[#list + 1] = entry
|
|
281
327
|
redis.call("hset", KEYS[1], "callbacks", cjson.encode(list))
|
|
282
328
|
return redis.call("hget", KEYS[1], ARGV[2])
|
|
283
329
|
LUA
|
|
284
330
|
|
|
285
331
|
# Ent Unique (§3): atomic compare-and-delete of a lock key. Replaces the
|
|
286
332
|
# two-command GET-then-DEL — between those calls the key can expire and a
|
|
287
|
-
# fresh
|
|
333
|
+
# fresh owner can grab it, and the bare DEL would then drop the new
|
|
288
334
|
# owner's lock. Shared by `Unique::ServerMiddleware#release` (normal
|
|
289
|
-
# success/start release)
|
|
290
|
-
# release)
|
|
291
|
-
#
|
|
292
|
-
#
|
|
335
|
+
# success/start release), `Unique::DEATH_HANDLER` (automatic-death
|
|
336
|
+
# release) and `Leader#release` (stepping down from the cluster lock)
|
|
337
|
+
# so those paths cannot drift.
|
|
338
|
+
# KEYS = [the lock key — unique:<sha256> | dear-leader]
|
|
339
|
+
# ARGV = [the owner that must still hold it — jid | <host>:<pid>:<nonce>]
|
|
293
340
|
# Returns 1 when the key was deleted, 0 otherwise.
|
|
294
341
|
RELEASE_IF_OWNER = <<~LUA
|
|
295
342
|
if redis.call("get", KEYS[1]) == ARGV[1] then
|
|
@@ -334,6 +381,41 @@ module Wurk
|
|
|
334
381
|
return removed
|
|
335
382
|
LUA
|
|
336
383
|
|
|
384
|
+
# Ent Periodic (§2): compare-and-swap the fire marks of `loops:<lid>`.
|
|
385
|
+
# The cron poller's leader gate is a cached read (`Component#leader?`,
|
|
386
|
+
# LEADER_CACHE_TTL_MS), so for a few seconds after a handover two
|
|
387
|
+
# processes both believe they lead and both reach the same due loop —
|
|
388
|
+
# HMGET → decide → enqueue → HSET then fires it twice. Claiming the slot
|
|
389
|
+
# atomically is what makes a tick fire exactly once; the loser gets 0 and
|
|
390
|
+
# enqueues nothing. Values written are the same decimal-epoch strings the
|
|
391
|
+
# Ruby path wrote, byte for byte — the dashboard reads these fields.
|
|
392
|
+
#
|
|
393
|
+
# `nf` present: the caller must still be looking at the mark it read
|
|
394
|
+
# (byte compare, so no parse can drift the token).
|
|
395
|
+
# `nf` absent: the caller derived the slot from `lf`, so refuse once `lf`
|
|
396
|
+
# has reached it — that is the exhausted-schedule case, where the winner
|
|
397
|
+
# cleared `nf` and a loser would otherwise re-fire the very same slot.
|
|
398
|
+
# KEYS = [loops:<lid>]
|
|
399
|
+
# ARGV = [expected `nf` (or the derived slot), new `lf`, new `nf` ('' → HDEL)]
|
|
400
|
+
# Returns 1 when this caller claimed the slot, 0 otherwise.
|
|
401
|
+
CRON_CLAIM_FIRE = <<~LUA
|
|
402
|
+
local key = KEYS[1]
|
|
403
|
+
local cur = redis.call("hget", key, "nf")
|
|
404
|
+
if cur and cur ~= "" then
|
|
405
|
+
if cur ~= ARGV[1] then return 0 end
|
|
406
|
+
else
|
|
407
|
+
local lf = tonumber(redis.call("hget", key, "lf") or "")
|
|
408
|
+
if lf and lf >= tonumber(ARGV[1]) then return 0 end
|
|
409
|
+
end
|
|
410
|
+
redis.call("hset", key, "lf", ARGV[2])
|
|
411
|
+
if ARGV[3] == "" then
|
|
412
|
+
redis.call("hdel", key, "nf")
|
|
413
|
+
else
|
|
414
|
+
redis.call("hset", key, "nf", ARGV[3])
|
|
415
|
+
end
|
|
416
|
+
return 1
|
|
417
|
+
LUA
|
|
418
|
+
|
|
337
419
|
# Limiter scripts live in `lib/wurk/lua/limiter_*.lua` — one file per
|
|
338
420
|
# type. Loaded at boot, the file's basename (minus `.lua`) becomes the
|
|
339
421
|
# SCRIPTS key as a symbol. Keeping them as separate files makes diffing
|
|
@@ -358,7 +440,8 @@ module Wurk
|
|
|
358
440
|
batch_append_callback: BATCH_APPEND_CALLBACK,
|
|
359
441
|
fast_delete_job: FAST_DELETE_JOB,
|
|
360
442
|
fast_delete_by_class: FAST_DELETE_BY_CLASS,
|
|
361
|
-
release_if_owner: RELEASE_IF_OWNER
|
|
443
|
+
release_if_owner: RELEASE_IF_OWNER,
|
|
444
|
+
cron_claim_fire: CRON_CLAIM_FIRE
|
|
362
445
|
}.merge(FILE_SCRIPTS).freeze
|
|
363
446
|
|
|
364
447
|
# SHA1 of each script source — matches what `SCRIPT LOAD` returns.
|
data/lib/wurk/manager.rb
CHANGED
|
@@ -24,12 +24,21 @@ module Wurk
|
|
|
24
24
|
|
|
25
25
|
attr_reader :workers, :capsule
|
|
26
26
|
|
|
27
|
-
|
|
27
|
+
# `shutdown:` is the owning Launcher's process-wide shutdown request,
|
|
28
|
+
# invoked when concurrency can no longer be held (see #processor_result).
|
|
29
|
+
# Injected rather than reached for: only the entry point knows how this
|
|
30
|
+
# process exits — a swarm child / standalone CLI re-delivers TERM to itself
|
|
31
|
+
# so the installed trap drives the normal drain, while embedded mode owns
|
|
32
|
+
# no traps and must stop the launcher in place. A Manager built without an
|
|
33
|
+
# owner (Sidekiq's `Manager.new(capsule)` arity, kept for the alias) has no
|
|
34
|
+
# route to take, so it only reports.
|
|
35
|
+
def initialize(capsule, shutdown: nil)
|
|
28
36
|
@config = @capsule = capsule
|
|
29
37
|
@count = capsule.concurrency
|
|
30
38
|
raise ArgumentError, "Concurrency of #{@count} is not supported" if @count < 1
|
|
31
39
|
|
|
32
40
|
@done = false
|
|
41
|
+
@shutdown = shutdown
|
|
33
42
|
@workers = Set.new
|
|
34
43
|
@plock = ::Mutex.new
|
|
35
44
|
@count.times do
|
|
@@ -86,8 +95,9 @@ module Wurk
|
|
|
86
95
|
# cleanly or via raised exception. Removes the dead processor from the
|
|
87
96
|
# pool and (unless we're already stopping) spawns a replacement so the
|
|
88
97
|
# capsule's concurrency stays constant. If the replacement itself can't be
|
|
89
|
-
# spawned,
|
|
90
|
-
#
|
|
98
|
+
# spawned, ask the owner to shut this process down (the swarm respawns it at
|
|
99
|
+
# full concurrency) rather than silently dropping a worker for the life of
|
|
100
|
+
# the process. Snapshot under @plock; start the replacement — a
|
|
91
101
|
# side effect — outside the lock.
|
|
92
102
|
def processor_result(processor, _reason = nil)
|
|
93
103
|
replacement = @plock.synchronize do
|
|
@@ -102,10 +112,15 @@ module Wurk
|
|
|
102
112
|
rescue StandardError => e
|
|
103
113
|
# Replacement spawn failed (e.g. ThreadError at the OS thread limit).
|
|
104
114
|
# Silently running one Processor short for the life of the process is
|
|
105
|
-
# invisible degradation
|
|
106
|
-
#
|
|
115
|
+
# invisible degradation, so take the process down (the swarm respawns it
|
|
116
|
+
# at full concurrency) — but through the owner's TERM path, never
|
|
117
|
+
# `Thread.main.raise`: that raise unwound straight past Manager#stop, so
|
|
118
|
+
# the in-flight UnitsOfWork were never bulk_requeued and each one waited
|
|
119
|
+
# out a full reaper interval, and in embedded mode it killed the host's
|
|
120
|
+
# main thread instead of just the worker. Non-blocking by contract — we
|
|
121
|
+
# are on the dying Processor's own thread, which the drain will kill.
|
|
107
122
|
@capsule.config.handle_exception(e, { context: 'Manager could not replace a dead Processor' })
|
|
108
|
-
|
|
123
|
+
@shutdown&.call
|
|
109
124
|
end
|
|
110
125
|
|
|
111
126
|
# Reached when the deadline expired with workers still busy. Atomically
|
|
@@ -146,14 +161,15 @@ module Wurk
|
|
|
146
161
|
@plock.synchronize { @workers.dup }
|
|
147
162
|
end
|
|
148
163
|
|
|
164
|
+
# Drained means "no processor thread is still running", not "the Set is
|
|
165
|
+
# empty": a processor removes itself from @workers from inside its own
|
|
166
|
+
# thread, so the two agree — except for a processor that was never started
|
|
167
|
+
# (a launcher that raised mid-boot and rolled back). That one holds no
|
|
168
|
+
# thread and nothing will ever remove it, so an emptiness test would make
|
|
169
|
+
# #stop poll out the entire shutdown deadline waiting on a worker that
|
|
170
|
+
# never ran.
|
|
149
171
|
def workers_empty?
|
|
150
|
-
@plock.synchronize { @workers.
|
|
151
|
-
end
|
|
152
|
-
|
|
153
|
-
# Seam over Thread.main: lets processor_result's replacement-failure crash
|
|
154
|
-
# be unit-tested against a controlled thread instead of the live runner.
|
|
155
|
-
def main_thread
|
|
156
|
-
Thread.main
|
|
172
|
+
@plock.synchronize { @workers.none? { |w| w.thread&.alive? } }
|
|
157
173
|
end
|
|
158
174
|
|
|
159
175
|
# Polls `condblock` until it returns true or the monotonic deadline
|
data/lib/wurk/metrics/history.rb
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require_relative '../middleware'
|
|
4
|
+
require_relative '../pool_checkout'
|
|
4
5
|
|
|
5
6
|
module Wurk
|
|
6
7
|
module Metrics
|
|
@@ -124,11 +125,11 @@ module Wurk
|
|
|
124
125
|
cfg.handle_exception(err, context: 'Wurk::Metrics::History')
|
|
125
126
|
end
|
|
126
127
|
|
|
127
|
-
def self.with_pool(pool, &)
|
|
128
|
+
def self.with_pool(pool, idempotent: false, &)
|
|
128
129
|
if pool
|
|
129
|
-
|
|
130
|
+
PoolCheckout.with(pool, idempotent, &)
|
|
130
131
|
else
|
|
131
|
-
Wurk.redis(&)
|
|
132
|
+
Wurk.redis(idempotent:, &)
|
|
132
133
|
end
|
|
133
134
|
end
|
|
134
135
|
private_class_method :with_pool
|
|
@@ -51,11 +51,23 @@ module Wurk
|
|
|
51
51
|
end
|
|
52
52
|
|
|
53
53
|
def start
|
|
54
|
-
@thread
|
|
54
|
+
return @thread if @thread
|
|
55
|
+
|
|
56
|
+
@timer.reset
|
|
57
|
+
@thread = safe_thread('queue-metrics') { @timer.run { tick } }
|
|
55
58
|
end
|
|
56
59
|
|
|
60
|
+
# Blocks until the thread is really gone: the launcher releases the
|
|
61
|
+
# cluster lock immediately after this returns, and a sample still in
|
|
62
|
+
# flight would write the same buckets as the next leader's first one.
|
|
63
|
+
#
|
|
64
|
+
# Cleared only on a confirmed join (Thread#join returns nil on timeout):
|
|
65
|
+
# a wedged thread must stay tracked so #start's guard returns it instead
|
|
66
|
+
# of calling @timer.reset, which would un-terminate the loop it is still
|
|
67
|
+
# inside and leave two threads HSETting the same buckets.
|
|
57
68
|
def terminate
|
|
58
69
|
@timer.terminate
|
|
70
|
+
@thread = nil if @thread&.join(TimerLoop::JOIN_TIMEOUT)
|
|
59
71
|
end
|
|
60
72
|
|
|
61
73
|
# Leader-gated: only the elected leader samples, so N workers don't each
|
data/lib/wurk/metrics/rollup.rb
CHANGED
|
@@ -59,11 +59,23 @@ module Wurk
|
|
|
59
59
|
end
|
|
60
60
|
|
|
61
61
|
def start
|
|
62
|
-
@thread
|
|
62
|
+
return @thread if @thread
|
|
63
|
+
|
|
64
|
+
@timer.reset
|
|
65
|
+
@thread = safe_thread('metrics-rollup') { @timer.run { tick } }
|
|
63
66
|
end
|
|
64
67
|
|
|
68
|
+
# Blocks until the thread is really gone: the launcher releases the
|
|
69
|
+
# cluster lock immediately after this returns, and a roll still in flight
|
|
70
|
+
# would write the same buckets as the next leader's first one.
|
|
71
|
+
#
|
|
72
|
+
# Cleared only on a confirmed join (Thread#join returns nil on timeout):
|
|
73
|
+
# a wedged thread must stay tracked so #start's guard returns it instead
|
|
74
|
+
# of calling @timer.reset, which would un-terminate the loop it is still
|
|
75
|
+
# inside and leave two threads HSETting the same buckets.
|
|
65
76
|
def terminate
|
|
66
77
|
@timer.terminate
|
|
78
|
+
@thread = nil if @thread&.join(TimerLoop::JOIN_TIMEOUT)
|
|
67
79
|
end
|
|
68
80
|
|
|
69
81
|
# Leader-gated: only the elected leader writes the cluster-total series,
|
|
@@ -8,14 +8,15 @@ module Wurk
|
|
|
8
8
|
module Middleware
|
|
9
9
|
# Server middleware. Catches `Wurk::Job::Interrupted` raised by an
|
|
10
10
|
# IterableJob mid-iteration (or any cooperatively-cancelled job),
|
|
11
|
-
# re-pushes the job to the
|
|
11
|
+
# re-pushes the job to the tail of its queue so it resumes from the
|
|
12
12
|
# persisted cursor, and raises `Wurk::JobRetry::Skip` so the retry
|
|
13
13
|
# layer treats this as a clean exit rather than an error.
|
|
14
14
|
#
|
|
15
|
-
# The re-push uses
|
|
16
|
-
# one to be fetched
|
|
17
|
-
#
|
|
18
|
-
#
|
|
15
|
+
# The re-push uses RPUSH (tail of queue) so the same job is the next
|
|
16
|
+
# one to be fetched: the fetcher's LMOVE pops from the RIGHT (tail),
|
|
17
|
+
# so this job is fetched ahead of fresh LPUSH'd enqueues. The job
|
|
18
|
+
# JSON is unchanged: cursor state lives in the `it-<jid>` HASH (see
|
|
19
|
+
# IterableJob persistence), not in the payload.
|
|
19
20
|
#
|
|
20
21
|
# Auto-registered at the top of the server chain when this file is
|
|
21
22
|
# required. Top-of-chain is important: a downstream middleware must
|
|
@@ -36,7 +37,7 @@ module Wurk
|
|
|
36
37
|
|
|
37
38
|
def repush(job, queue)
|
|
38
39
|
payload = Wurk.dump_json(job)
|
|
39
|
-
redis_pool.with { |conn| conn.call('
|
|
40
|
+
redis_pool.with { |conn| conn.call('RPUSH', "queue:#{queue}", payload) }
|
|
40
41
|
end
|
|
41
42
|
end
|
|
42
43
|
end
|
|
@@ -19,15 +19,22 @@ module Wurk
|
|
|
19
19
|
# Counter key TTL is wire-compat with Sidekiq Pro — third-party tooling
|
|
20
20
|
# that watches `super_fetch:recovered:*` expects 72h.
|
|
21
21
|
#
|
|
22
|
-
# No server-middleware registration:
|
|
23
|
-
#
|
|
24
|
-
#
|
|
25
|
-
#
|
|
22
|
+
# No server-middleware registration: the producer is Reaper#drain, which
|
|
23
|
+
# drives the lifecycle directly via `track!(payload, queue:)` on each
|
|
24
|
+
# orphan it moves back to the public queue. The consumer is the ACK —
|
|
25
|
+
# `Fetcher::Reliable::UnitOfWork#acknowledge` pipelines {clear_in} next to
|
|
26
|
+
# its LREM, so an attempt that finished starts the next one at zero
|
|
27
|
+
# without costing a round trip of its own.
|
|
26
28
|
module PoisonPill
|
|
27
29
|
RECOVERY_THRESHOLD = 3
|
|
28
30
|
RECOVERY_TTL = 72 * 60 * 60
|
|
29
31
|
KEY_PREFIX = 'super_fetch:recovered:'
|
|
30
|
-
|
|
32
|
+
|
|
33
|
+
# Handed to death handlers (and therefore to `:death` batch callbacks)
|
|
34
|
+
# as the cause when a recovered job is killed as a poison pill. Never
|
|
35
|
+
# raised: nothing is on a stack here — the reaper kills the job from the
|
|
36
|
+
# outside, on behalf of the workers it took down.
|
|
37
|
+
class Poisoned < ::StandardError; end
|
|
31
38
|
|
|
32
39
|
# The `pill` handed to a Pro `super_fetch! { |jobstr, pill| }` recovery
|
|
33
40
|
# callback on the kill path. Responds to .jid/.klass/.count/.queue so a
|
|
@@ -62,7 +69,7 @@ module Wurk
|
|
|
62
69
|
|
|
63
70
|
jid = job['jid']
|
|
64
71
|
klass = job['class']
|
|
65
|
-
|
|
72
|
+
emit('jobs.recovered.fetch', klass, queue)
|
|
66
73
|
|
|
67
74
|
count = bump_counter(jid) if jid && !jid.empty?
|
|
68
75
|
if count && count >= RECOVERY_THRESHOLD
|
|
@@ -80,15 +87,43 @@ module Wurk
|
|
|
80
87
|
def recovery_count(jid)
|
|
81
88
|
return 0 if jid.nil? || jid.to_s.empty?
|
|
82
89
|
|
|
83
|
-
Wurk.redis { |conn| conn.call('GET',
|
|
90
|
+
Wurk.redis { |conn| conn.call('GET', counter_key(jid)) }.to_i
|
|
84
91
|
end
|
|
85
92
|
|
|
86
|
-
# Resets the counter for a jid
|
|
87
|
-
#
|
|
93
|
+
# Resets the counter for a jid. The hot path uses {clear_in} instead —
|
|
94
|
+
# this is the standalone form for the API/dashboard and for callers with
|
|
95
|
+
# no pipeline of their own.
|
|
88
96
|
def clear!(jid)
|
|
89
97
|
return if jid.nil? || jid.to_s.empty?
|
|
90
98
|
|
|
91
|
-
Wurk.redis { |conn| conn.call('DEL',
|
|
99
|
+
Wurk.redis { |conn| conn.call('DEL', counter_key(jid)) }
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
# Queue the counter reset onto a pipeline the caller already has open.
|
|
103
|
+
#
|
|
104
|
+
# The ACK is the reset point. An attempt that got as far as acking —
|
|
105
|
+
# returned, or raised and booked its retry — is proof the job did not
|
|
106
|
+
# take its worker down, and that is the only thing this counter
|
|
107
|
+
# measures. Without the reset, three reclaims caused by three unrelated
|
|
108
|
+
# crashes inside 72h dead-set a job that has been completing all along
|
|
109
|
+
# (jids do come back: a UI/API retry re-pushes the same one, as does any
|
|
110
|
+
# client that supplies its own). Riding a round trip the ACK already
|
|
111
|
+
# makes is what keeps that free for the jobs — nearly all of them — that
|
|
112
|
+
# were never reclaimed at all.
|
|
113
|
+
# Guards the blank jid for the same reason {clear!} and {recovery_count}
|
|
114
|
+
# do, rather than relying on the one caller to do it: `counter_key(nil)`
|
|
115
|
+
# is the bare KEY_PREFIX, so an unguarded DEL here would delete a key
|
|
116
|
+
# that is shared rather than per-job. The caller's own check stays --
|
|
117
|
+
# it also skips queueing the command at all -- but the invariant belongs
|
|
118
|
+
# on the method that builds the key.
|
|
119
|
+
def clear_in(pipe, jid)
|
|
120
|
+
return if jid.nil? || jid.to_s.empty?
|
|
121
|
+
|
|
122
|
+
pipe.call('DEL', counter_key(jid))
|
|
123
|
+
end
|
|
124
|
+
|
|
125
|
+
def counter_key(jid)
|
|
126
|
+
"#{KEY_PREFIX}#{jid}"
|
|
92
127
|
end
|
|
93
128
|
|
|
94
129
|
# Register a callback fired when a poison pill is detected. Callbacks
|
|
@@ -113,19 +148,21 @@ module Wurk
|
|
|
113
148
|
# ---- internals --------------------------------------------------
|
|
114
149
|
|
|
115
150
|
def parse(payload)
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
nil
|
|
123
|
-
end
|
|
124
|
-
end
|
|
151
|
+
return payload if payload.is_a?(::Hash)
|
|
152
|
+
return nil unless payload.is_a?(::String)
|
|
153
|
+
|
|
154
|
+
Wurk.load_json(payload)
|
|
155
|
+
rescue ::JSON::ParserError
|
|
156
|
+
nil
|
|
125
157
|
end
|
|
126
158
|
|
|
159
|
+
# No apply-safety claim: INCR is additive, so a block replayed after a
|
|
160
|
+
# lost reply bumps twice and can carry a healthy job past
|
|
161
|
+
# RECOVERY_THRESHOLD into the dead set. Raising instead only leaves the
|
|
162
|
+
# count short — Reaper#drain rescues, and the job is already back on its
|
|
163
|
+
# public queue, so it just misses one poison check.
|
|
127
164
|
def bump_counter(jid)
|
|
128
|
-
key =
|
|
165
|
+
key = counter_key(jid)
|
|
129
166
|
Wurk.redis do |conn|
|
|
130
167
|
count = conn.call('INCR', key).to_i
|
|
131
168
|
conn.call('EXPIRE', key, RECOVERY_TTL)
|
|
@@ -133,25 +170,29 @@ module Wurk
|
|
|
133
170
|
end
|
|
134
171
|
end
|
|
135
172
|
|
|
136
|
-
def
|
|
173
|
+
def emit(metric, klass, queue)
|
|
137
174
|
tags = []
|
|
138
175
|
tags << "class:#{klass}" if klass
|
|
139
176
|
tags << "queue:#{queue}" if queue
|
|
140
|
-
Wurk::Metrics::Statsd.increment(
|
|
177
|
+
Wurk::Metrics::Statsd.increment(metric, tags: tags.empty? ? nil : tags)
|
|
141
178
|
end
|
|
142
179
|
|
|
180
|
+
# Death handlers fire (`notify_failure` defaults to true). A poison kill
|
|
181
|
+
# is a death like any other exhaustion: suppressing it strands every
|
|
182
|
+
# batch that owns the job — Batch::DeathHandler is a death handler, so a
|
|
183
|
+
# silent kill leaves the batch's pending count stuck forever and neither
|
|
184
|
+
# `:death` nor `:complete` ever runs. Pro's spec is silent here
|
|
185
|
+
# (docs/target/sidekiq-pro.md §12); recorded in docs/idea/parity-divergences.md.
|
|
143
186
|
def mark_poison(payload, job, queue:, count:)
|
|
144
|
-
|
|
187
|
+
emit('jobs.poison', job['class'], queue)
|
|
145
188
|
json = payload.is_a?(String) ? payload : Wurk.dump_json(job)
|
|
146
|
-
Wurk::DeadSet.new.kill(json,
|
|
189
|
+
Wurk::DeadSet.new.kill(json, ex: poisoned_error(job['class'], count))
|
|
147
190
|
fire_callbacks(jid: job['jid'], klass: job['class'], count: count, queue: queue)
|
|
148
191
|
end
|
|
149
192
|
|
|
150
|
-
def
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
tags << "queue:#{queue}" if queue
|
|
154
|
-
Wurk::Metrics::Statsd.increment('jobs.poison', tags: tags.empty? ? nil : tags)
|
|
193
|
+
def poisoned_error(klass, count)
|
|
194
|
+
message = "#{klass || 'job'} was recovered #{count} times without completing"
|
|
195
|
+
Poisoned.new(message).tap { |e| e.set_backtrace(caller) }
|
|
155
196
|
end
|
|
156
197
|
|
|
157
198
|
def fire_callbacks(pill)
|
data/lib/wurk/middleware.rb
CHANGED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative 'redis_pool'
|
|
4
|
+
|
|
5
|
+
module Wurk
|
|
6
|
+
# The seam every `#redis`-style wrapper checks out through.
|
|
7
|
+
#
|
|
8
|
+
# `idempotent:` is a {Wurk::RedisPool} concept — it re-enables block replay
|
|
9
|
+
# after a command may already have applied server-side (see RedisPool#with).
|
|
10
|
+
# The object behind a wrapper is not always ours, though: the Sidekiq surface
|
|
11
|
+
# Wurk drops into lets a host supply any ConnectionPool-shaped object
|
|
12
|
+
# (`Sidekiq::Client#redis_pool=`, `Sidekiq::Web.redis_pool=`, the `pool:`
|
|
13
|
+
# kwarg on Stats::History / Deploy / Leader / Profiler / Metrics::History),
|
|
14
|
+
# and a bare `def with; yield conn; end` decorator — the shape people reach
|
|
15
|
+
# for to count or trace round trips — is a legal implementation of it. Handing
|
|
16
|
+
# that an unknown keyword is an ArgumentError, so the claim goes only to a pool
|
|
17
|
+
# that understands it.
|
|
18
|
+
#
|
|
19
|
+
# Dropping the claim is always sound: the pool falls back to the conservative
|
|
20
|
+
# no-replay default, which is exactly what a foreign pool did before the
|
|
21
|
+
# keyword existed. Apply-safety is an optimization, never a correctness input.
|
|
22
|
+
module PoolCheckout
|
|
23
|
+
def self.with(pool, idempotent, &)
|
|
24
|
+
return pool.with(&) unless idempotent && pool.is_a?(RedisPool)
|
|
25
|
+
|
|
26
|
+
pool.with(idempotent: true, &)
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
end
|