wurk 1.3.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +1 -0
  3. data/lib/wurk/batch/callbacks.rb +82 -12
  4. data/lib/wurk/batch/death_handler.rb +7 -4
  5. data/lib/wurk/batch/server_middleware.rb +1 -1
  6. data/lib/wurk/batch.rb +121 -15
  7. data/lib/wurk/capsule.rb +5 -4
  8. data/lib/wurk/cli.rb +48 -14
  9. data/lib/wurk/client/buffered.rb +193 -43
  10. data/lib/wurk/client.rb +87 -14
  11. data/lib/wurk/compat.rb +1 -1
  12. data/lib/wurk/component.rb +2 -2
  13. data/lib/wurk/configuration.rb +10 -2
  14. data/lib/wurk/cron.rb +94 -37
  15. data/lib/wurk/deploy.rb +5 -3
  16. data/lib/wurk/embedded.rb +13 -0
  17. data/lib/wurk/fetcher/reaper.rb +113 -56
  18. data/lib/wurk/fetcher/reliable.rb +62 -9
  19. data/lib/wurk/heartbeat.rb +22 -10
  20. data/lib/wurk/history.rb +13 -1
  21. data/lib/wurk/launcher.rb +133 -66
  22. data/lib/wurk/leader.rb +29 -10
  23. data/lib/wurk/limiter/base.rb +8 -10
  24. data/lib/wurk/limiter/bucket.rb +1 -1
  25. data/lib/wurk/limiter/concurrent.rb +27 -22
  26. data/lib/wurk/limiter/window.rb +13 -11
  27. data/lib/wurk/limiter.rb +7 -4
  28. data/lib/wurk/lua.rb +97 -14
  29. data/lib/wurk/manager.rb +29 -13
  30. data/lib/wurk/metrics/history.rb +4 -3
  31. data/lib/wurk/metrics/queue_rollup.rb +13 -1
  32. data/lib/wurk/metrics/rollup.rb +13 -1
  33. data/lib/wurk/middleware/interrupt_handler.rb +7 -6
  34. data/lib/wurk/middleware/poison_pill.rb +70 -29
  35. data/lib/wurk/middleware.rb +2 -2
  36. data/lib/wurk/pool_checkout.rb +29 -0
  37. data/lib/wurk/process_set.rb +10 -5
  38. data/lib/wurk/processor.rb +6 -0
  39. data/lib/wurk/profiler.rb +3 -2
  40. data/lib/wurk/queue.rb +10 -7
  41. data/lib/wurk/rails_boot.rb +38 -7
  42. data/lib/wurk/redis_client_adapter.rb +48 -4
  43. data/lib/wurk/redis_options.rb +142 -0
  44. data/lib/wurk/redis_pool.rb +102 -39
  45. data/lib/wurk/scheduled.rb +30 -2
  46. data/lib/wurk/stats.rb +14 -9
  47. data/lib/wurk/swarm/child_boot.rb +12 -0
  48. data/lib/wurk/swarm.rb +174 -33
  49. data/lib/wurk/timer_loop.rb +14 -0
  50. data/lib/wurk/version.rb +1 -1
  51. data/lib/wurk/web/enterprise.rb +58 -6
  52. data/lib/wurk/web/extension.rb +1 -1
  53. data/lib/wurk/web/search.rb +5 -3
  54. data/lib/wurk.rb +53 -2
  55. data/vendor/assets/dashboard/assets/{BatchDetail-YRymNsrB.js → BatchDetail-OmC5NPgw.js} +1 -1
  56. data/vendor/assets/dashboard/assets/{Batches-HY4hHdQU.js → Batches-CIpai7St.js} +1 -1
  57. data/vendor/assets/dashboard/assets/{Busy-FCEN1Bpx.js → Busy-A_kwSR6Q.js} +1 -1
  58. data/vendor/assets/dashboard/assets/{Cron-DO3J2zcp.js → Cron-BG7HTqlp.js} +1 -1
  59. data/vendor/assets/dashboard/assets/{Dashboard-B9rOrkzk.js → Dashboard-A_ToqHoo.js} +1 -1
  60. data/vendor/assets/dashboard/assets/{Dead-Bi4GGk9a.js → Dead-8J21jMyK.js} +1 -1
  61. data/vendor/assets/dashboard/assets/Extension-B4Q9FIQu.js +1 -0
  62. data/vendor/assets/dashboard/assets/{FilterBox-IJkHYpdm.js → FilterBox-Fh_Ae7UW.js} +1 -1
  63. data/vendor/assets/dashboard/assets/{JobDetailModal-DS1ypyoc.js → JobDetailModal-Ceng0PMB.js} +1 -1
  64. data/vendor/assets/dashboard/assets/{Limiters-Nz7UbNeJ.js → Limiters-CruDWvNZ.js} +1 -1
  65. data/vendor/assets/dashboard/assets/{Metrics-BBTDxcaE.js → Metrics-CIT7VCoN.js} +1 -1
  66. data/vendor/assets/dashboard/assets/Modal-CN3rdKA_.js +1 -0
  67. data/vendor/assets/dashboard/assets/{Queues-D9PH_THs.js → Queues-D86FYohJ.js} +1 -1
  68. data/vendor/assets/dashboard/assets/{Retries-CAKzDgYG.js → Retries-Bz1O1D-i.js} +1 -1
  69. data/vendor/assets/dashboard/assets/{Scheduled-DTYw1X8S.js → Scheduled-B6h2akTu.js} +1 -1
  70. data/vendor/assets/dashboard/assets/{Search-c4vFxDG_.js → Search-OOu22e5s.js} +1 -1
  71. data/vendor/assets/dashboard/assets/index-BdiUEDXX.css +1 -0
  72. data/vendor/assets/dashboard/assets/index-D_lSDwKw.js +141 -0
  73. data/vendor/assets/dashboard/assets/{useResetPageOnEmpty-B_FsMah6.js → useResetPageOnEmpty-dVPGEWzn.js} +1 -1
  74. data/vendor/assets/dashboard/index.html +2 -2
  75. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  76. metadata +22 -20
  77. data/vendor/assets/dashboard/assets/Extension-BSv8ddW_.js +0 -1
  78. data/vendor/assets/dashboard/assets/Modal-Crrsu64-.js +0 -1
  79. data/vendor/assets/dashboard/assets/index-BxjdeuOa.css +0 -1
  80. data/vendor/assets/dashboard/assets/index-DQu7WY9y.js +0 -141
data/lib/wurk/limiter.rb CHANGED
@@ -4,6 +4,7 @@ require 'json'
4
4
  require 'digest'
5
5
  require 'securerandom'
6
6
  require_relative 'lua'
7
+ require_relative 'pool_checkout'
7
8
 
8
9
  module Wurk
9
10
  # Sidekiq Enterprise rate limiters: concurrent, bucket, window, leaky,
@@ -24,7 +25,10 @@ module Wurk
24
25
  # Layout (one file per type under `lib/wurk/limiter/`):
25
26
  # * `Limiter::Base` owns the metadata write (lmtr:{name}) + the global
26
27
  # `lmtr-list` registration so the Web UI can list every limiter, and
27
- # the uniform `status` shape.
28
+ # the uniform `status` shape. Membership is added once per metadata
29
+ # write and swept back out by `Web::Enterprise::Limits.list` once the
30
+ # metadata expires — the two halves of keeping `lmtr-list` bounded
31
+ # under interpolated names.
28
32
  # * Per-type subclasses (Concurrent / Bucket / Window / Leaky / Points)
29
33
  # own their acquire/wait loop. Each delegates the atomic step to a
30
34
  # Lua script in `lib/wurk/lua/limiter_*.lua`.
@@ -146,9 +150,8 @@ module Wurk
146
150
  # Redis access: caller-supplied pool (Limiter.configure.redis = …) wins,
147
151
  # else fall back to the default Wurk pool. This is the same hierarchy
148
152
  # Sidekiq Ent documents — dedicated rate-limiter pool is opt-in.
149
- def redis(&)
150
- pool = config.pool || Wurk.redis_pool
151
- pool.with(&)
153
+ def redis(idempotent: false, &)
154
+ PoolCheckout.with(config.pool || Wurk.redis_pool, idempotent, &)
152
155
  end
153
156
 
154
157
  # `ZRANGE key 0 0 WITHSCORES` yields a single [member, score] pair, but the
data/lib/wurk/lua.rb CHANGED
@@ -131,8 +131,16 @@ module Wurk
131
131
  # success after the dead job is manually retried to success). The
132
132
  # `b-<bid>-death` notify dedup key is untouched, so `:death` cannot
133
133
  # re-fire.
134
+ #
135
+ # The two EXPIRE NX calls are what keep the batch bounded: `b-<bid>-jids` is
136
+ # born here, and only `Batch#ensure_first_flush!` ever stamped a TTL — on the
137
+ # hash alone — so an abandoned or invalidated batch leaked its live-jid set
138
+ # forever. A late re-push (retry, scheduled promotion) can also resurrect an
139
+ # already-expired `b-<bid>` through the HINCRBYs above, again TTL-less. NX
140
+ # stamps only keys that currently have no TTL, so a running batch's clock and
141
+ # the shorter post-success `linger` window (Callbacks#apply_linger) both win.
134
142
  # KEYS = [b-<bid>, b-<bid>-jids, queue_list, queues_set, b-<bid>-died, dead-batches]
135
- # ARGV = [queue_name, jid, job_json, bid]
143
+ # ARGV = [queue_name, jid, job_json, bid, expiry_seconds]
136
144
  # Returns 1.
137
145
  BATCH_PUSH = <<~LUA
138
146
  if redis.call("srem", KEYS[5], ARGV[2]) == 1 then
@@ -147,6 +155,8 @@ module Wurk
147
155
  redis.call("hincrby", KEYS[1], "pending", 1)
148
156
  end
149
157
  end
158
+ redis.call("expire", KEYS[1], ARGV[5], "NX")
159
+ redis.call("expire", KEYS[2], ARGV[5], "NX")
150
160
  redis.call("sadd", KEYS[4], ARGV[1])
151
161
  redis.call("lpush", KEYS[3], ARGV[3])
152
162
  return 1
@@ -165,14 +175,19 @@ module Wurk
165
175
  # it, the re-push routes through BATCH_PUSH, whose guard finds the jid already
166
176
  # live (SADD == 0) → pure LPUSH, no recount. So registration happens exactly
167
177
  # once, here, at enqueue.
178
+ #
179
+ # The other path that can create `b-<bid>-jids`, so it carries the same NX
180
+ # expiry stamp as BATCH_PUSH.
168
181
  # KEYS = [schedule, b-<bid>, b-<bid>-jids]
169
- # ARGV = [at_score, job_json, jid]
182
+ # ARGV = [at_score, job_json, jid, expiry_seconds]
170
183
  # Returns 1.
171
184
  BATCH_SCHEDULE = <<~LUA
172
185
  if redis.call("sadd", KEYS[3], ARGV[3]) == 1 then
173
186
  redis.call("hincrby", KEYS[2], "total", 1)
174
187
  redis.call("hincrby", KEYS[2], "pending", 1)
175
188
  end
189
+ redis.call("expire", KEYS[2], ARGV[4], "NX")
190
+ redis.call("expire", KEYS[3], ARGV[4], "NX")
176
191
  redis.call("zadd", KEYS[1], ARGV[1], ARGV[2])
177
192
  return 1
178
193
  LUA
@@ -208,13 +223,18 @@ module Wurk
208
223
  # currently in a failing/retrying state. Re-failures of the same jid are
209
224
  # idempotent. Cleared by BATCH_ACK_SUCCESS (retry passed) or
210
225
  # BATCH_ACK_COMPLETE (job died). Spec §2.5, §2.8.
226
+ #
227
+ # `b-<bid>-failed` is born here, so it carries the same NX expiry stamp as
228
+ # BATCH_PUSH.
211
229
  # KEYS = [b-<bid>, b-<bid>-failed]
212
- # ARGV = [jid]
230
+ # ARGV = [jid, expiry_seconds]
213
231
  # Returns 1.
214
232
  BATCH_ACK_FAILED = <<~LUA
215
233
  if redis.call("sadd", KEYS[2], ARGV[1]) == 1 then
216
234
  redis.call("hincrby", KEYS[1], "failures", 1)
217
235
  end
236
+ redis.call("expire", KEYS[1], ARGV[2], "NX")
237
+ redis.call("expire", KEYS[2], ARGV[2], "NX")
218
238
  return 1
219
239
  LUA
220
240
 
@@ -224,8 +244,11 @@ module Wurk
224
244
  # live jids so the batch can fire `:complete` even with terminally failed
225
245
  # jobs. `b-<bid>-failed` holds only currently-retrying jids; `b-<bid>-died`
226
246
  # holds terminally-dead ones (spec §2.8 — the two sets are distinct).
247
+ #
248
+ # `b-<bid>-died` is born here, so it carries the same NX expiry stamp as
249
+ # BATCH_PUSH.
227
250
  # KEYS = [b-<bid>, b-<bid>-jids, b-<bid>-died, b-<bid>-failed]
228
- # ARGV = [jid]
251
+ # ARGV = [jid, expiry_seconds]
229
252
  # Returns [live_jids_remaining, died_count, first_death]. `first_death`
230
253
  # is 1 the first time *any* jid is SADDed into the died set, 0 thereafter
231
254
  # — caller uses it to fire `:death` exactly once per batch.
@@ -236,6 +259,8 @@ module Wurk
236
259
  redis.call("hincrby", KEYS[1], "failures", -1)
237
260
  end
238
261
  local died_added = redis.call("sadd", KEYS[3], ARGV[1])
262
+ redis.call("expire", KEYS[1], ARGV[2], "NX")
263
+ redis.call("expire", KEYS[3], ARGV[2], "NX")
239
264
  local first_death = 0
240
265
  if was_pre_existing_death == 0 and died_added == 1 then
241
266
  first_death = 1
@@ -262,10 +287,21 @@ module Wurk
262
287
  # same reopened batch cannot lose each other's writes. Refuses to write
263
288
  # when the batch hash is gone — resurrecting a bare hash would create a
264
289
  # batch that can never fire anything.
290
+ #
291
+ # Appends are deduped and capped. Reopening the same batch inside every
292
+ # job and re-registering its callback is a common shape, and each append
293
+ # pays a decode + encode of the whole array: unbounded that is O(N^2) Lua
294
+ # work on the way in and N identical callback jobs at fire time. An
295
+ # identical triple is therefore a no-op — the registration is already
296
+ # satisfied by the entry that is there — and past the cap the append is
297
+ # refused rather than allowed to grow the hash field without limit.
298
+ # Both sides of the comparison are normalised through cjson so an entry
299
+ # written by Ruby's `to_json` at first flush matches one appended here.
265
300
  # KEYS = [b-<bid>]
266
- # ARGV = [callback triple JSON, event name]
267
- # Returns -1 when the batch hash does not exist; otherwise the event's
268
- # fired flag ("1", or nil when it has not fired yet).
301
+ # ARGV = [callback triple JSON, event name, max callbacks]
302
+ # Returns -1 when the batch hash does not exist and -2 when the cap
303
+ # refused the append; otherwise the event's fired flag ("1", or nil when
304
+ # it has not fired yet).
269
305
  BATCH_APPEND_CALLBACK = <<~LUA
270
306
  if redis.call("exists", KEYS[1]) == 0 then
271
307
  return -1
@@ -277,19 +313,30 @@ module Wurk
277
313
  else
278
314
  list = {}
279
315
  end
280
- list[#list + 1] = cjson.decode(ARGV[1])
316
+ local entry = cjson.decode(ARGV[1])
317
+ local encoded = cjson.encode(entry)
318
+ for i = 1, #list do
319
+ if cjson.encode(list[i]) == encoded then
320
+ return redis.call("hget", KEYS[1], ARGV[2])
321
+ end
322
+ end
323
+ if #list >= tonumber(ARGV[3]) then
324
+ return -2
325
+ end
326
+ list[#list + 1] = entry
281
327
  redis.call("hset", KEYS[1], "callbacks", cjson.encode(list))
282
328
  return redis.call("hget", KEYS[1], ARGV[2])
283
329
  LUA
284
330
 
285
331
  # Ent Unique (§3): atomic compare-and-delete of a lock key. Replaces the
286
332
  # two-command GET-then-DEL — between those calls the key can expire and a
287
- # fresh enqueue can grab it, and the bare DEL would then drop the new
333
+ # fresh owner can grab it, and the bare DEL would then drop the new
288
334
  # owner's lock. Shared by `Unique::ServerMiddleware#release` (normal
289
- # success/start release) and `Unique::DEATH_HANDLER` (automatic-death
290
- # release) so the two paths cannot drift.
291
- # KEYS = [unique:<sha256>]
292
- # ARGV = [owning jid]
335
+ # success/start release), `Unique::DEATH_HANDLER` (automatic-death
336
+ # release) and `Leader#release` (stepping down from the cluster lock)
337
+ # so those paths cannot drift.
338
+ # KEYS = [the lock key — unique:<sha256> | dear-leader]
339
+ # ARGV = [the owner that must still hold it — jid | <host>:<pid>:<nonce>]
293
340
  # Returns 1 when the key was deleted, 0 otherwise.
294
341
  RELEASE_IF_OWNER = <<~LUA
295
342
  if redis.call("get", KEYS[1]) == ARGV[1] then
@@ -334,6 +381,41 @@ module Wurk
334
381
  return removed
335
382
  LUA
336
383
 
384
+ # Ent Periodic (§2): compare-and-swap the fire marks of `loops:<lid>`.
385
+ # The cron poller's leader gate is a cached read (`Component#leader?`,
386
+ # LEADER_CACHE_TTL_MS), so for a few seconds after a handover two
387
+ # processes both believe they lead and both reach the same due loop —
388
+ # HMGET → decide → enqueue → HSET then fires it twice. Claiming the slot
389
+ # atomically is what makes a tick fire exactly once; the loser gets 0 and
390
+ # enqueues nothing. Values written are the same decimal-epoch strings the
391
+ # Ruby path wrote, byte for byte — the dashboard reads these fields.
392
+ #
393
+ # `nf` present: the caller must still be looking at the mark it read
394
+ # (byte compare, so no parse can drift the token).
395
+ # `nf` absent: the caller derived the slot from `lf`, so refuse once `lf`
396
+ # has reached it — that is the exhausted-schedule case, where the winner
397
+ # cleared `nf` and a loser would otherwise re-fire the very same slot.
398
+ # KEYS = [loops:<lid>]
399
+ # ARGV = [expected `nf` (or the derived slot), new `lf`, new `nf` ('' → HDEL)]
400
+ # Returns 1 when this caller claimed the slot, 0 otherwise.
401
+ CRON_CLAIM_FIRE = <<~LUA
402
+ local key = KEYS[1]
403
+ local cur = redis.call("hget", key, "nf")
404
+ if cur and cur ~= "" then
405
+ if cur ~= ARGV[1] then return 0 end
406
+ else
407
+ local lf = tonumber(redis.call("hget", key, "lf") or "")
408
+ if lf and lf >= tonumber(ARGV[1]) then return 0 end
409
+ end
410
+ redis.call("hset", key, "lf", ARGV[2])
411
+ if ARGV[3] == "" then
412
+ redis.call("hdel", key, "nf")
413
+ else
414
+ redis.call("hset", key, "nf", ARGV[3])
415
+ end
416
+ return 1
417
+ LUA
418
+
337
419
  # Limiter scripts live in `lib/wurk/lua/limiter_*.lua` — one file per
338
420
  # type. Loaded at boot, the file's basename (minus `.lua`) becomes the
339
421
  # SCRIPTS key as a symbol. Keeping them as separate files makes diffing
@@ -358,7 +440,8 @@ module Wurk
358
440
  batch_append_callback: BATCH_APPEND_CALLBACK,
359
441
  fast_delete_job: FAST_DELETE_JOB,
360
442
  fast_delete_by_class: FAST_DELETE_BY_CLASS,
361
- release_if_owner: RELEASE_IF_OWNER
443
+ release_if_owner: RELEASE_IF_OWNER,
444
+ cron_claim_fire: CRON_CLAIM_FIRE
362
445
  }.merge(FILE_SCRIPTS).freeze
363
446
 
364
447
  # SHA1 of each script source — matches what `SCRIPT LOAD` returns.
data/lib/wurk/manager.rb CHANGED
@@ -24,12 +24,21 @@ module Wurk
24
24
 
25
25
  attr_reader :workers, :capsule
26
26
 
27
- def initialize(capsule)
27
+ # `shutdown:` is the owning Launcher's process-wide shutdown request,
28
+ # invoked when concurrency can no longer be held (see #processor_result).
29
+ # Injected rather than reached for: only the entry point knows how this
30
+ # process exits — a swarm child / standalone CLI re-delivers TERM to itself
31
+ # so the installed trap drives the normal drain, while embedded mode owns
32
+ # no traps and must stop the launcher in place. A Manager built without an
33
+ # owner (Sidekiq's `Manager.new(capsule)` arity, kept for the alias) has no
34
+ # route to take, so it only reports.
35
+ def initialize(capsule, shutdown: nil)
28
36
  @config = @capsule = capsule
29
37
  @count = capsule.concurrency
30
38
  raise ArgumentError, "Concurrency of #{@count} is not supported" if @count < 1
31
39
 
32
40
  @done = false
41
+ @shutdown = shutdown
33
42
  @workers = Set.new
34
43
  @plock = ::Mutex.new
35
44
  @count.times do
@@ -86,8 +95,9 @@ module Wurk
86
95
  # cleanly or via raised exception. Removes the dead processor from the
87
96
  # pool and (unless we're already stopping) spawns a replacement so the
88
97
  # capsule's concurrency stays constant. If the replacement itself can't be
89
- # spawned, crash the child (the swarm respawns it) rather than silently
90
- # dropping concurrency. Snapshot under @plock; start the replacement — a
98
+ # spawned, ask the owner to shut this process down (the swarm respawns it at
99
+ # full concurrency) rather than silently dropping a worker for the life of
100
+ # the process. Snapshot under @plock; start the replacement — a
91
101
  # side effect — outside the lock.
92
102
  def processor_result(processor, _reason = nil)
93
103
  replacement = @plock.synchronize do
@@ -102,10 +112,15 @@ module Wurk
102
112
  rescue StandardError => e
103
113
  # Replacement spawn failed (e.g. ThreadError at the OS thread limit).
104
114
  # Silently running one Processor short for the life of the process is
105
- # invisible degradation; instead report and crash the child on the main
106
- # thread so the swarm respawns it at full concurrency (plan 02 §6).
115
+ # invisible degradation, so take the process down (the swarm respawns it
116
+ # at full concurrency) — but through the owner's TERM path, never
117
+ # `Thread.main.raise`: that raise unwound straight past Manager#stop, so
118
+ # the in-flight UnitsOfWork were never bulk_requeued and each one waited
119
+ # out a full reaper interval, and in embedded mode it killed the host's
120
+ # main thread instead of just the worker. Non-blocking by contract — we
121
+ # are on the dying Processor's own thread, which the drain will kill.
107
122
  @capsule.config.handle_exception(e, { context: 'Manager could not replace a dead Processor' })
108
- main_thread.raise(e)
123
+ @shutdown&.call
109
124
  end
110
125
 
111
126
  # Reached when the deadline expired with workers still busy. Atomically
@@ -146,14 +161,15 @@ module Wurk
146
161
  @plock.synchronize { @workers.dup }
147
162
  end
148
163
 
164
+ # Drained means "no processor thread is still running", not "the Set is
165
+ # empty": a processor removes itself from @workers from inside its own
166
+ # thread, so the two agree — except for a processor that was never started
167
+ # (a launcher that raised mid-boot and rolled back). That one holds no
168
+ # thread and nothing will ever remove it, so an emptiness test would make
169
+ # #stop poll out the entire shutdown deadline waiting on a worker that
170
+ # never ran.
149
171
  def workers_empty?
150
- @plock.synchronize { @workers.empty? }
151
- end
152
-
153
- # Seam over Thread.main: lets processor_result's replacement-failure crash
154
- # be unit-tested against a controlled thread instead of the live runner.
155
- def main_thread
156
- Thread.main
172
+ @plock.synchronize { @workers.none? { |w| w.thread&.alive? } }
157
173
  end
158
174
 
159
175
  # Polls `condblock` until it returns true or the monotonic deadline
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require_relative '../middleware'
4
+ require_relative '../pool_checkout'
4
5
 
5
6
  module Wurk
6
7
  module Metrics
@@ -124,11 +125,11 @@ module Wurk
124
125
  cfg.handle_exception(err, context: 'Wurk::Metrics::History')
125
126
  end
126
127
 
127
- def self.with_pool(pool, &)
128
+ def self.with_pool(pool, idempotent: false, &)
128
129
  if pool
129
- pool.with(&)
130
+ PoolCheckout.with(pool, idempotent, &)
130
131
  else
131
- Wurk.redis(&)
132
+ Wurk.redis(idempotent:, &)
132
133
  end
133
134
  end
134
135
  private_class_method :with_pool
@@ -51,11 +51,23 @@ module Wurk
51
51
  end
52
52
 
53
53
  def start
54
- @thread ||= safe_thread('queue-metrics') { @timer.run { tick } } # rubocop:disable Naming/MemoizedInstanceVariableName
54
+ return @thread if @thread
55
+
56
+ @timer.reset
57
+ @thread = safe_thread('queue-metrics') { @timer.run { tick } }
55
58
  end
56
59
 
60
+ # Blocks until the thread is really gone: the launcher releases the
61
+ # cluster lock immediately after this returns, and a sample still in
62
+ # flight would write the same buckets as the next leader's first one.
63
+ #
64
+ # Cleared only on a confirmed join (Thread#join returns nil on timeout):
65
+ # a wedged thread must stay tracked so #start's guard returns it instead
66
+ # of calling @timer.reset, which would un-terminate the loop it is still
67
+ # inside and leave two threads HSETting the same buckets.
57
68
  def terminate
58
69
  @timer.terminate
70
+ @thread = nil if @thread&.join(TimerLoop::JOIN_TIMEOUT)
59
71
  end
60
72
 
61
73
  # Leader-gated: only the elected leader samples, so N workers don't each
@@ -59,11 +59,23 @@ module Wurk
59
59
  end
60
60
 
61
61
  def start
62
- @thread ||= safe_thread('metrics-rollup') { @timer.run { tick } } # rubocop:disable Naming/MemoizedInstanceVariableName
62
+ return @thread if @thread
63
+
64
+ @timer.reset
65
+ @thread = safe_thread('metrics-rollup') { @timer.run { tick } }
63
66
  end
64
67
 
68
+ # Blocks until the thread is really gone: the launcher releases the
69
+ # cluster lock immediately after this returns, and a roll still in flight
70
+ # would write the same buckets as the next leader's first one.
71
+ #
72
+ # Cleared only on a confirmed join (Thread#join returns nil on timeout):
73
+ # a wedged thread must stay tracked so #start's guard returns it instead
74
+ # of calling @timer.reset, which would un-terminate the loop it is still
75
+ # inside and leave two threads HSETting the same buckets.
65
76
  def terminate
66
77
  @timer.terminate
78
+ @thread = nil if @thread&.join(TimerLoop::JOIN_TIMEOUT)
67
79
  end
68
80
 
69
81
  # Leader-gated: only the elected leader writes the cluster-total series,
@@ -8,14 +8,15 @@ module Wurk
8
8
  module Middleware
9
9
  # Server middleware. Catches `Wurk::Job::Interrupted` raised by an
10
10
  # IterableJob mid-iteration (or any cooperatively-cancelled job),
11
- # re-pushes the job to the head of its queue so it resumes from the
11
+ # re-pushes the job to the tail of its queue so it resumes from the
12
12
  # persisted cursor, and raises `Wurk::JobRetry::Skip` so the retry
13
13
  # layer treats this as a clean exit rather than an error.
14
14
  #
15
- # The re-push uses LPUSH (head of queue) so the same job is the next
16
- # one to be fetched after restart. The job JSON is unchanged: cursor
17
- # state lives in the `it-<jid>` HASH (see IterableJob persistence),
18
- # not in the payload.
15
+ # The re-push uses RPUSH (tail of queue) so the same job is the next
16
+ # one to be fetched: the fetcher's LMOVE pops from the RIGHT (tail),
17
+ # so this job is fetched ahead of fresh LPUSH'd enqueues. The job
18
+ # JSON is unchanged: cursor state lives in the `it-<jid>` HASH (see
19
+ # IterableJob persistence), not in the payload.
19
20
  #
20
21
  # Auto-registered at the top of the server chain when this file is
21
22
  # required. Top-of-chain is important: a downstream middleware must
@@ -36,7 +37,7 @@ module Wurk
36
37
 
37
38
  def repush(job, queue)
38
39
  payload = Wurk.dump_json(job)
39
- redis_pool.with { |conn| conn.call('LPUSH', "queue:#{queue}", payload) }
40
+ redis_pool.with { |conn| conn.call('RPUSH', "queue:#{queue}", payload) }
40
41
  end
41
42
  end
42
43
  end
@@ -19,15 +19,22 @@ module Wurk
19
19
  # Counter key TTL is wire-compat with Sidekiq Pro — third-party tooling
20
20
  # that watches `super_fetch:recovered:*` expects 72h.
21
21
  #
22
- # No server-middleware registration: callers are reaper / bulk_requeue
23
- # paths that drive the lifecycle directly via `track!(payload, queue:)`.
24
- # When that integration lands, it just calls `PoisonPill.track!` on each
25
- # orphan it about to RPUSH back to the public queue.
22
+ # No server-middleware registration: the producer is Reaper#drain, which
23
+ # drives the lifecycle directly via `track!(payload, queue:)` on each
24
+ # orphan it moves back to the public queue. The consumer is the ACK —
25
+ # `Fetcher::Reliable::UnitOfWork#acknowledge` pipelines {clear_in} next to
26
+ # its LREM, so an attempt that finished starts the next one at zero
27
+ # without costing a round trip of its own.
26
28
  module PoisonPill
27
29
  RECOVERY_THRESHOLD = 3
28
30
  RECOVERY_TTL = 72 * 60 * 60
29
31
  KEY_PREFIX = 'super_fetch:recovered:'
30
- DEAD_RECORD_LIMIT = 100
32
+
33
+ # Handed to death handlers (and therefore to `:death` batch callbacks)
34
+ # as the cause when a recovered job is killed as a poison pill. Never
35
+ # raised: nothing is on a stack here — the reaper kills the job from the
36
+ # outside, on behalf of the workers it took down.
37
+ class Poisoned < ::StandardError; end
31
38
 
32
39
  # The `pill` handed to a Pro `super_fetch! { |jobstr, pill| }` recovery
33
40
  # callback on the kill path. Responds to .jid/.klass/.count/.queue so a
@@ -62,7 +69,7 @@ module Wurk
62
69
 
63
70
  jid = job['jid']
64
71
  klass = job['class']
65
- emit_recovered_fetch(klass, queue)
72
+ emit('jobs.recovered.fetch', klass, queue)
66
73
 
67
74
  count = bump_counter(jid) if jid && !jid.empty?
68
75
  if count && count >= RECOVERY_THRESHOLD
@@ -80,15 +87,43 @@ module Wurk
80
87
  def recovery_count(jid)
81
88
  return 0 if jid.nil? || jid.to_s.empty?
82
89
 
83
- Wurk.redis { |conn| conn.call('GET', "#{KEY_PREFIX}#{jid}") }.to_i
90
+ Wurk.redis { |conn| conn.call('GET', counter_key(jid)) }.to_i
84
91
  end
85
92
 
86
- # Resets the counter for a jid — call after a successful perform so a
87
- # job that recovered twice and then completed doesn't accumulate state.
93
+ # Resets the counter for a jid. The hot path uses {clear_in} instead —
94
+ # this is the standalone form for the API/dashboard and for callers with
95
+ # no pipeline of their own.
88
96
  def clear!(jid)
89
97
  return if jid.nil? || jid.to_s.empty?
90
98
 
91
- Wurk.redis { |conn| conn.call('DEL', "#{KEY_PREFIX}#{jid}") }
99
+ Wurk.redis { |conn| conn.call('DEL', counter_key(jid)) }
100
+ end
101
+
102
+ # Queue the counter reset onto a pipeline the caller already has open.
103
+ #
104
+ # The ACK is the reset point. An attempt that got as far as acking —
105
+ # returned, or raised and booked its retry — is proof the job did not
106
+ # take its worker down, and that is the only thing this counter
107
+ # measures. Without the reset, three reclaims caused by three unrelated
108
+ # crashes inside 72h dead-set a job that has been completing all along
109
+ # (jids do come back: a UI/API retry re-pushes the same one, as does any
110
+ # client that supplies its own). Riding a round trip the ACK already
111
+ # makes is what keeps that free for the jobs — nearly all of them — that
112
+ # were never reclaimed at all.
113
+ # Guards the blank jid for the same reason {clear!} and {recovery_count}
114
+ # do, rather than relying on the one caller to do it: `counter_key(nil)`
115
+ # is the bare KEY_PREFIX, so an unguarded DEL here would delete a key
116
+ # that is shared rather than per-job. The caller's own check stays --
117
+ # it also skips queueing the command at all -- but the invariant belongs
118
+ # on the method that builds the key.
119
+ def clear_in(pipe, jid)
120
+ return if jid.nil? || jid.to_s.empty?
121
+
122
+ pipe.call('DEL', counter_key(jid))
123
+ end
124
+
125
+ def counter_key(jid)
126
+ "#{KEY_PREFIX}#{jid}"
92
127
  end
93
128
 
94
129
  # Register a callback fired when a poison pill is detected. Callbacks
@@ -113,19 +148,21 @@ module Wurk
113
148
  # ---- internals --------------------------------------------------
114
149
 
115
150
  def parse(payload)
116
- case payload
117
- when Hash then payload
118
- when String
119
- begin
120
- Wurk.load_json(payload)
121
- rescue ::JSON::ParserError
122
- nil
123
- end
124
- end
151
+ return payload if payload.is_a?(::Hash)
152
+ return nil unless payload.is_a?(::String)
153
+
154
+ Wurk.load_json(payload)
155
+ rescue ::JSON::ParserError
156
+ nil
125
157
  end
126
158
 
159
+ # No apply-safety claim: INCR is additive, so a block replayed after a
160
+ # lost reply bumps twice and can carry a healthy job past
161
+ # RECOVERY_THRESHOLD into the dead set. Raising instead only leaves the
162
+ # count short — Reaper#drain rescues, and the job is already back on its
163
+ # public queue, so it just misses one poison check.
127
164
  def bump_counter(jid)
128
- key = "#{KEY_PREFIX}#{jid}"
165
+ key = counter_key(jid)
129
166
  Wurk.redis do |conn|
130
167
  count = conn.call('INCR', key).to_i
131
168
  conn.call('EXPIRE', key, RECOVERY_TTL)
@@ -133,25 +170,29 @@ module Wurk
133
170
  end
134
171
  end
135
172
 
136
- def emit_recovered_fetch(klass, queue)
173
+ def emit(metric, klass, queue)
137
174
  tags = []
138
175
  tags << "class:#{klass}" if klass
139
176
  tags << "queue:#{queue}" if queue
140
- Wurk::Metrics::Statsd.increment('jobs.recovered.fetch', tags: tags.empty? ? nil : tags)
177
+ Wurk::Metrics::Statsd.increment(metric, tags: tags.empty? ? nil : tags)
141
178
  end
142
179
 
180
+ # Death handlers fire (`notify_failure` defaults to true). A poison kill
181
+ # is a death like any other exhaustion: suppressing it strands every
182
+ # batch that owns the job — Batch::DeathHandler is a death handler, so a
183
+ # silent kill leaves the batch's pending count stuck forever and neither
184
+ # `:death` nor `:complete` ever runs. Pro's spec is silent here
185
+ # (docs/target/sidekiq-pro.md §12); recorded in docs/idea/parity-divergences.md.
143
186
  def mark_poison(payload, job, queue:, count:)
144
- emit_poison(job['class'], queue)
187
+ emit('jobs.poison', job['class'], queue)
145
188
  json = payload.is_a?(String) ? payload : Wurk.dump_json(job)
146
- Wurk::DeadSet.new.kill(json, notify_failure: false)
189
+ Wurk::DeadSet.new.kill(json, ex: poisoned_error(job['class'], count))
147
190
  fire_callbacks(jid: job['jid'], klass: job['class'], count: count, queue: queue)
148
191
  end
149
192
 
150
- def emit_poison(klass, queue)
151
- tags = []
152
- tags << "class:#{klass}" if klass
153
- tags << "queue:#{queue}" if queue
154
- Wurk::Metrics::Statsd.increment('jobs.poison', tags: tags.empty? ? nil : tags)
193
+ def poisoned_error(klass, count)
194
+ message = "#{klass || 'job'} was recovered #{count} times without completing"
195
+ Poisoned.new(message).tap { |e| e.set_backtrace(caller) }
155
196
  end
156
197
 
157
198
  def fire_callbacks(pill)
@@ -24,8 +24,8 @@ module Wurk
24
24
  config.logger
25
25
  end
26
26
 
27
- def redis(&)
28
- config.redis(&)
27
+ def redis(idempotent: false, &)
28
+ config.redis(idempotent:, &)
29
29
  end
30
30
  end
31
31
 
@@ -0,0 +1,29 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative 'redis_pool'
4
+
5
+ module Wurk
6
+ # The seam every `#redis`-style wrapper checks out through.
7
+ #
8
+ # `idempotent:` is a {Wurk::RedisPool} concept — it re-enables block replay
9
+ # after a command may already have applied server-side (see RedisPool#with).
10
+ # The object behind a wrapper is not always ours, though: the Sidekiq surface
11
+ # Wurk drops into lets a host supply any ConnectionPool-shaped object
12
+ # (`Sidekiq::Client#redis_pool=`, `Sidekiq::Web.redis_pool=`, the `pool:`
13
+ # kwarg on Stats::History / Deploy / Leader / Profiler / Metrics::History),
14
+ # and a bare `def with; yield conn; end` decorator — the shape people reach
15
+ # for to count or trace round trips — is a legal implementation of it. Handing
16
+ # that an unknown keyword is an ArgumentError, so the claim goes only to a pool
17
+ # that understands it.
18
+ #
19
+ # Dropping the claim is always sound: the pool falls back to the conservative
20
+ # no-replay default, which is exactly what a foreign pool did before the
21
+ # keyword existed. Apply-safety is an optimization, never a correctness input.
22
+ module PoolCheckout
23
+ def self.with(pool, idempotent, &)
24
+ return pool.with(&) unless idempotent && pool.is_a?(RedisPool)
25
+
26
+ pool.with(idempotent: true, &)
27
+ end
28
+ end
29
+ end